{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n    <h1 align=\"center\" style=\"color:darkgreen;\">Tabular Playground Series - Jul 2022</h1>  \n</div>\n\n<div>\n    <h1 align=\"center\" style=\"color:darkgreen;\">C l u s t e r i n g - 2/3</h1>\n</div>\n\n<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"markdown","source":"#### In this notebook, we improve the results of our first notebook.\n\n#### For iteration, we use sklego.mixture, which has good speed.\n\n#### Thanks to @karlcini for introducing \"sklego\" recently.","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"import warnings # suppress warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:32.632465Z","iopub.execute_input":"2022-07-23T10:08:32.632893Z","iopub.status.idle":"2022-07-23T10:08:32.637732Z","shell.execute_reply.started":"2022-07-23T10:08:32.63285Z","shell.execute_reply":"2022-07-23T10:08:32.636857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport random\n\nimport numpy as np \nimport pandas as pd\nimport seaborn as sns\n\nfrom tqdm import tqdm\nfrom scipy import stats\nfrom pathlib import Path\n\nimport matplotlib.pyplot as plt\nimport plotly.figure_factory as ff\nimport plotly.express as px\n%matplotlib inline\n!ls ../input/*","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:32.643944Z","iopub.execute_input":"2022-07-23T10:08:32.644356Z","iopub.status.idle":"2022-07-23T10:08:33.429272Z","shell.execute_reply.started":"2022-07-23T10:08:32.644315Z","shell.execute_reply":"2022-07-23T10:08:33.427881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install sklego","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-23T10:08:33.432061Z","iopub.execute_input":"2022-07-23T10:08:33.432433Z","iopub.status.idle":"2022-07-23T10:08:44.658438Z","shell.execute_reply.started":"2022-07-23T10:08:33.432399Z","shell.execute_reply":"2022-07-23T10:08:44.657387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklego.mixture import BayesianGMMClassifier\nfrom sklearn.mixture import BayesianGaussianMixture\n\nfrom sklearn.preprocessing import MinMaxScaler, PowerTransformer, StandardScaler, RobustScaler, LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:44.659745Z","iopub.execute_input":"2022-07-23T10:08:44.660079Z","iopub.status.idle":"2022-07-23T10:08:44.666432Z","shell.execute_reply.started":"2022-07-23T10:08:44.660049Z","shell.execute_reply":"2022-07-23T10:08:44.665257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div>\n    <h1 align=\"center\" style=\"color:darkgreen;\">Load Data & Preprocessing</h1>\n</div>\n\n<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"DATA = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\nDATA","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:44.669946Z","iopub.execute_input":"2022-07-23T10:08:44.670358Z","iopub.status.idle":"2022-07-23T10:08:45.585785Z","shell.execute_reply.started":"2022-07-23T10:08:44.670315Z","shell.execute_reply":"2022-07-23T10:08:45.584609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\nSAMPLE","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.587581Z","iopub.execute_input":"2022-07-23T10:08:45.588288Z","iopub.status.idle":"2022-07-23T10:08:45.617217Z","shell.execute_reply.started":"2022-07-23T10:08:45.58824Z","shell.execute_reply":"2022-07-23T10:08:45.616074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Missing Values:')\nDATA.isna().sum().sum()\n# DATA.isna().sum()","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-23T10:08:45.61897Z","iopub.execute_input":"2022-07-23T10:08:45.619757Z","iopub.status.idle":"2022-07-23T10:08:45.635218Z","shell.execute_reply.started":"2022-07-23T10:08:45.619708Z","shell.execute_reply":"2022-07-23T10:08:45.634061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Duplicates in Data:')\nDATA.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.63678Z","iopub.execute_input":"2022-07-23T10:08:45.637447Z","iopub.status.idle":"2022-07-23T10:08:45.85874Z","shell.execute_reply.started":"2022-07-23T10:08:45.637414Z","shell.execute_reply":"2022-07-23T10:08:45.857382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"df = DATA.copy()\ndf.drop(\"id\", axis=1, inplace=True)\ncols = list(df.columns)\n\ndf.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.860258Z","iopub.execute_input":"2022-07-23T10:08:45.861238Z","iopub.status.idle":"2022-07-23T10:08:45.883654Z","shell.execute_reply.started":"2022-07-23T10:08:45.861198Z","shell.execute_reply":"2022-07-23T10:08:45.882305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Discrete Data\ncols_dis = df.columns[df.dtypes == 'int']\n# cols_dis = [x for (x, y) in df.dtypes.items() if y == 'int']\ndf_dis = df[cols_dis]\nprint(cols_dis)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.885185Z","iopub.execute_input":"2022-07-23T10:08:45.885641Z","iopub.status.idle":"2022-07-23T10:08:45.895975Z","shell.execute_reply.started":"2022-07-23T10:08:45.885599Z","shell.execute_reply":"2022-07-23T10:08:45.894541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.select_dtypes('int').nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.901319Z","iopub.execute_input":"2022-07-23T10:08:45.902045Z","iopub.status.idle":"2022-07-23T10:08:45.919618Z","shell.execute_reply.started":"2022-07-23T10:08:45.901997Z","shell.execute_reply":"2022-07-23T10:08:45.918295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:darkred;\">|?|</span>**\n\n#### <span style=\"color:darkgreen;\">Numerous tests showed that probably none of the Discrete Data columns are categorical data. For example, they can be the count of a value or... expressed as an integer number.</span>","metadata":{}},{"cell_type":"code","source":"# Continuous Data\ncols_con = df.columns[df.dtypes == 'float']\n# cols_con = [x for (x, y) in df.dtypes.items() if y == 'float']\ndf_con = df[cols_con]\nprint(cols_con)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.921485Z","iopub.execute_input":"2022-07-23T10:08:45.922726Z","iopub.status.idle":"2022-07-23T10:08:45.934721Z","shell.execute_reply.started":"2022-07-23T10:08:45.922679Z","shell.execute_reply":"2022-07-23T10:08:45.93328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.select_dtypes('float').nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:45.936942Z","iopub.execute_input":"2022-07-23T10:08:45.937799Z","iopub.status.idle":"2022-07-23T10:08:46.05581Z","shell.execute_reply.started":"2022-07-23T10:08:45.937751Z","shell.execute_reply":"2022-07-23T10:08:46.054589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = df.copy()\ndfs = PowerTransformer().fit_transform(dfs)\ndfs = pd.DataFrame(dfs, columns=cols)\n\nbgm = BayesianGaussianMixture(n_components=7,covariance_type='full',random_state=123)\nprd = bgm.fit_predict(dfs)\nprd","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:08:46.057148Z","iopub.execute_input":"2022-07-23T10:08:46.057557Z","iopub.status.idle":"2022-07-23T10:09:44.592696Z","shell.execute_reply.started":"2022-07-23T10:08:46.057497Z","shell.execute_reply":"2022-07-23T10:09:44.591574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('seaborn-whitegrid')\nplt.figure(figsize=(18,8), facecolor='lightblue')\nplt.title(f'\\nF e a t u r e s\\n')\nfor i in range(bgm.means_.shape[0]):\n    plt.scatter(np.arange(dfs.shape[1]), bgm.means_[i], s=30)\nplt.xticks(ticks=np.arange(dfs.shape[1]), labels=dfs.columns)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:09:44.594657Z","iopub.execute_input":"2022-07-23T10:09:44.59543Z","iopub.status.idle":"2022-07-23T10:09:44.95962Z","shell.execute_reply.started":"2022-07-23T10:09:44.595386Z","shell.execute_reply":"2022-07-23T10:09:44.95834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:darkred;\">|?|</span>**\n\n#### <span style=\"color:darkgreen;\">Preliminary studies in a number of public notebooks show that the bottom columns contain more useful data</span>","metadata":{}},{"cell_type":"code","source":"cols_select = []\nalpha = 0.05\n\nfor col in cols:\n    _, p_value = stats.shapiro(df[col])\n    \n    if (p_value <= alpha): \n        cols_select.append(col)       \nprint(cols_select)        ","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:09:44.960993Z","iopub.execute_input":"2022-07-23T10:09:44.961339Z","iopub.status.idle":"2022-07-23T10:09:45.224968Z","shell.execute_reply.started":"2022-07-23T10:09:44.961308Z","shell.execute_reply":"2022-07-23T10:09:45.223732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"dff = DATA[cols_select]\ndff","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:09:45.226461Z","iopub.execute_input":"2022-07-23T10:09:45.226836Z","iopub.status.idle":"2022-07-23T10:09:45.253346Z","shell.execute_reply.started":"2022-07-23T10:09:45.226802Z","shell.execute_reply":"2022-07-23T10:09:45.25241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:darkgreen;\">Scaling</span>","metadata":{}},{"cell_type":"code","source":"dffs = dff.copy()\ndffs = PowerTransformer().fit_transform(dffs)\ndffs = pd.DataFrame(dffs, columns=cols_select)\ndffs","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:09:45.254302Z","iopub.execute_input":"2022-07-23T10:09:45.254653Z","iopub.status.idle":"2022-07-23T10:09:47.039233Z","shell.execute_reply.started":"2022-07-23T10:09:45.254622Z","shell.execute_reply":"2022-07-23T10:09:47.038168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div>\n    <h1 align=\"center\" style=\"color:darkgreen;\">Pseudo-Supervised Clustering</h1>\n</div>\n\n<div class=\"alert alert-success\">  \n</div>","metadata":{"_kg_hide-output":true}},{"cell_type":"code","source":"N = 12 # The number of iteration\n\nsub_prime = pd.read_csv('../input/1-3-tps22jul-pseudo-supervised-clustering/submission.csv', index_col=[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T10:09:47.046936Z","iopub.execute_input":"2022-07-23T10:09:47.047234Z","iopub.status.idle":"2022-07-23T10:09:47.096568Z","shell.execute_reply.started":"2022-07-23T10:09:47.047206Z","shell.execute_reply":"2022-07-23T10:09:47.095546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **<span style=\"color:darkred;\">|?|</span>**\n\n#### <span style=\"color:darkgreen;\">The number of iterations, the number N, is determined by trial and error. By changing the input data, this number must be determined again.</span>","metadata":{}},{"cell_type":"code","source":"X = np.array(dffs)\ny = np.array(sub_prime)\n\nbgm = BayesianGMMClassifier(n_components=7, random_state=123, tol=0.001, max_iter=200, n_init=3, verbose=0)\n\nfor n in tqdm(range(N)):\n    bgm.fit(X,y)\n    pred = bgm.predict(X)\n    y = pred","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div>\n    <h1 align=\"center\" style=\"color:darkgreen;\">Submission</h1>\n</div>\n\n<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"sub = SAMPLE.copy()\nsub['Predicted'] = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:12:51.376659Z","iopub.execute_input":"2022-07-23T09:12:51.377148Z","iopub.status.idle":"2022-07-23T09:12:51.384422Z","shell.execute_reply.started":"2022-07-23T09:12:51.377101Z","shell.execute_reply":"2022-07-23T09:12:51.383067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist_data = [sub_prime.iloc[:, 0], pred]  \ngroup_labels = ['Sub_Prime', 'Submission']\n  \nfig = ff.create_distplot(hist_data, group_labels, bin_size=.2, show_hist=False, show_rug=False) \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:19:19.867406Z","iopub.execute_input":"2022-07-23T09:19:19.867812Z","iopub.status.idle":"2022-07-23T09:19:21.365837Z","shell.execute_reply.started":"2022-07-23T09:19:19.867779Z","shell.execute_reply":"2022-07-23T09:19:21.364112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)\n!ls","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:12:52.875332Z","iopub.execute_input":"2022-07-23T09:12:52.87615Z","iopub.status.idle":"2022-07-23T09:12:53.802713Z","shell.execute_reply.started":"2022-07-23T09:12:52.876118Z","shell.execute_reply":"2022-07-23T09:12:53.801434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n    <h1 align=\"center\" style=\"color:darkgreen;\">Good Luck</h1>  \n</div>","metadata":{}}]}