{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T12:32:03.348242Z","iopub.execute_input":"2022-07-12T12:32:03.348692Z","iopub.status.idle":"2022-07-12T12:32:03.390062Z","shell.execute_reply.started":"2022-07-12T12:32:03.348597Z","shell.execute_reply":"2022-07-12T12:32:03.388753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv(r'/kaggle/input/tabular-playground-series-jul-2022/data.csv')\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:03.658930Z","iopub.execute_input":"2022-07-12T12:32:03.660105Z","iopub.status.idle":"2022-07-12T12:32:04.726799Z","shell.execute_reply.started":"2022-07-12T12:32:03.660062Z","shell.execute_reply":"2022-07-12T12:32:04.725530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1=df.drop(['id'],axis=1)\ndf1.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:04.729850Z","iopub.execute_input":"2022-07-12T12:32:04.731190Z","iopub.status.idle":"2022-07-12T12:32:04.779119Z","shell.execute_reply.started":"2022-07-12T12:32:04.731138Z","shell.execute_reply":"2022-07-12T12:32:04.777553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2=pd.astype","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:04.781199Z","iopub.execute_input":"2022-07-12T12:32:04.781621Z","iopub.status.idle":"2022-07-12T12:32:04.788593Z","shell.execute_reply.started":"2022-07-12T12:32:04.781578Z","shell.execute_reply":"2022-07-12T12:32:04.786801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, RobustScaler, PowerTransformer,QuantileTransformer","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:04.791245Z","iopub.execute_input":"2022-07-12T12:32:04.792715Z","iopub.status.idle":"2022-07-12T12:32:05.297820Z","shell.execute_reply.started":"2022-07-12T12:32:04.792671Z","shell.execute_reply":"2022-07-12T12:32:05.296496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_scaled = PowerTransformer().fit_transform(df1)\n# X_scaled = PowerTransformer().fit_transform(X_scaled)\n\nX_scaled = pd.DataFrame(X_scaled)\nX_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:05.300791Z","iopub.execute_input":"2022-07-12T12:32:05.301606Z","iopub.status.idle":"2022-07-12T12:32:09.146527Z","shell.execute_reply.started":"2022-07-12T12:32:05.301560Z","shell.execute_reply":"2022-07-12T12:32:09.145044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.mixture import GaussianMixture, BayesianGaussianMixture\n# from sklearn.metrics import davies_bouldin_score\n# gmm = BayesianGaussianMixture(n_components = 9)\n# preds = gmm.fit_predict(X_scaled)\n# output = pd.DataFrame({'Id': df.id, 'Predicted': preds})\n# output\n# output.to_csv('submissionBGM9.csv', index=False)\n# print(\"Your submission was successfully saved!\")\n# scoreGM = davies_bouldin_score(X_scaled, preds)\n# scoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:09.150073Z","iopub.execute_input":"2022-07-12T12:32:09.150872Z","iopub.status.idle":"2022-07-12T12:32:09.157025Z","shell.execute_reply.started":"2022-07-12T12:32:09.150828Z","shell.execute_reply":"2022-07-12T12:32:09.155215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(20,20))\nsns.heatmap(df1.corr(),annot=True)\n#non of them are closely co related 7-13,22-28","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:09.158403Z","iopub.execute_input":"2022-07-12T12:32:09.160080Z","iopub.status.idle":"2022-07-12T12:32:14.349159Z","shell.execute_reply.started":"2022-07-12T12:32:09.160032Z","shell.execute_reply":"2022-07-12T12:32:14.348062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(df.f_01,df.f_02)\nplt.subplots(1,1)\nsns.scatterplot(X_scaled[0],X_scaled[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:14.351998Z","iopub.execute_input":"2022-07-12T12:32:14.359224Z","iopub.status.idle":"2022-07-12T12:32:15.234662Z","shell.execute_reply.started":"2022-07-12T12:32:14.359177Z","shell.execute_reply":"2022-07-12T12:32:15.233482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2=df1*10\ndf2=df2.astype('int32')\ndf2=df2/10\nfor i in df2:\n    df2[i]=(df2[i]-df2[i].min())/(df2[i].max()-df2[i].min())\ndf2","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:15.236298Z","iopub.execute_input":"2022-07-12T12:32:15.238351Z","iopub.status.idle":"2022-07-12T12:32:15.380947Z","shell.execute_reply.started":"2022-07-12T12:32:15.238290Z","shell.execute_reply":"2022-07-12T12:32:15.379766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2p = PowerTransformer().fit_transform(df2)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:15.383071Z","iopub.execute_input":"2022-07-12T12:32:15.383483Z","iopub.status.idle":"2022-07-12T12:32:18.077912Z","shell.execute_reply.started":"2022-07-12T12:32:15.383440Z","shell.execute_reply":"2022-07-12T12:32:18.076752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(df2p[12],df2p[9])\nplt.subplots(1,1)\nsns.scatterplot(X_scaled[12],X_scaled[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:18.079397Z","iopub.execute_input":"2022-07-12T12:32:18.080006Z","iopub.status.idle":"2022-07-12T12:32:18.705687Z","shell.execute_reply.started":"2022-07-12T12:32:18.079942Z","shell.execute_reply":"2022-07-12T12:32:18.704363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# StandardScaler, RobustScaler, PowerTransformer,QuantileTransformer\nss = StandardScaler().fit_transform(df1)\nss = PowerTransformer().fit_transform(ss)\nss = pd.DataFrame(ss)\nsns.scatterplot(ss[12],ss[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:18.707330Z","iopub.execute_input":"2022-07-12T12:32:18.708156Z","iopub.status.idle":"2022-07-12T12:32:22.983432Z","shell.execute_reply.started":"2022-07-12T12:32:18.708124Z","shell.execute_reply":"2022-07-12T12:32:22.982038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# StandardScaler, RobustScaler, PowerTransformer,QuantileTransformer\n# ss = StandardScaler().fit_transform(df1)\nssq = QuantileTransformer().fit_transform(ss)\nssq = pd.DataFrame(ssq)\nsns.scatterplot(ssq[12],ssq[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:22.985077Z","iopub.execute_input":"2022-07-12T12:32:22.985735Z","iopub.status.idle":"2022-07-12T12:32:24.062749Z","shell.execute_reply.started":"2022-07-12T12:32:22.985678Z","shell.execute_reply":"2022-07-12T12:32:24.061424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# StandardScaler, RobustScaler, PowerTransformer,QuantileTransformer\n# ss = StandardScaler().fit_transform(df1)\nqt = QuantileTransformer().fit_transform(df1)\nqt = pd.DataFrame(qt)\nsns.scatterplot(qt[12],qt[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:24.067807Z","iopub.execute_input":"2022-07-12T12:32:24.068561Z","iopub.status.idle":"2022-07-12T12:32:25.128840Z","shell.execute_reply.started":"2022-07-12T12:32:24.068515Z","shell.execute_reply":"2022-07-12T12:32:25.127457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qt2 = QuantileTransformer().fit_transform(df2)\nqt2 = pd.DataFrame(qt2)\nsns.scatterplot(qt2[12],qt2[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:25.131316Z","iopub.execute_input":"2022-07-12T12:32:25.132246Z","iopub.status.idle":"2022-07-12T12:32:26.092437Z","shell.execute_reply.started":"2022-07-12T12:32:25.132203Z","shell.execute_reply":"2022-07-12T12:32:26.091070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# StandardScaler, RobustScaler, PowerTransformer,QuantileTransformer\nrqt = RobustScaler().fit_transform(df1)\nrqt = QuantileTransformer().fit_transform(df1)\nrqt = pd.DataFrame(rqt)\nsns.scatterplot(rqt[12],rqt[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:26.094523Z","iopub.execute_input":"2022-07-12T12:32:26.095452Z","iopub.status.idle":"2022-07-12T12:32:27.273143Z","shell.execute_reply.started":"2022-07-12T12:32:26.095393Z","shell.execute_reply":"2022-07-12T12:32:27.271773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rsp = StandardScaler().fit_transform(df1)\nrsp = PowerTransformer().fit_transform(rsp)\nrsp = pd.DataFrame(rsp)\nsns.scatterplot(ss[12],ss[9])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:27.275059Z","iopub.execute_input":"2022-07-12T12:32:27.275807Z","iopub.status.idle":"2022-07-12T12:32:31.608481Z","shell.execute_reply.started":"2022-07-12T12:32:27.275762Z","shell.execute_reply":"2022-07-12T12:32:31.607035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.metrics import davies_bouldin_score\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(df2p)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submissiondf2p.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(df2p, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:32:31.610481Z","iopub.execute_input":"2022-07-12T12:32:31.610913Z","iopub.status.idle":"2022-07-12T12:33:43.071876Z","shell.execute_reply.started":"2022-07-12T12:32:31.610868Z","shell.execute_reply":"2022-07-12T12:33:43.069049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\n\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(ss)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submission_ss.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(ss, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:33:43.073614Z","iopub.execute_input":"2022-07-12T12:33:43.074070Z","iopub.status.idle":"2022-07-12T12:35:11.347159Z","shell.execute_reply.started":"2022-07-12T12:33:43.074027Z","shell.execute_reply":"2022-07-12T12:35:11.344089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\n\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(ssq)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submission_ssq.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(ssq, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:35:11.349045Z","iopub.execute_input":"2022-07-12T12:35:11.349509Z","iopub.status.idle":"2022-07-12T12:36:45.728266Z","shell.execute_reply.started":"2022-07-12T12:35:11.349466Z","shell.execute_reply":"2022-07-12T12:36:45.724539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\n\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(qt)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submission_qt.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(qt, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:37:45.310272Z","iopub.execute_input":"2022-07-12T12:37:45.310642Z","iopub.status.idle":"2022-07-12T12:39:14.573776Z","shell.execute_reply.started":"2022-07-12T12:37:45.310612Z","shell.execute_reply":"2022-07-12T12:39:14.570829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\n\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(qt2)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submission_qt2.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(qt2, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:39:41.755319Z","iopub.execute_input":"2022-07-12T12:39:41.755725Z","iopub.status.idle":"2022-07-12T12:41:10.563718Z","shell.execute_reply.started":"2022-07-12T12:39:41.755685Z","shell.execute_reply":"2022-07-12T12:41:10.562046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\n\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(rqt)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submission_rqt.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(rqt, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:43:12.565672Z","iopub.execute_input":"2022-07-12T12:43:12.566104Z","iopub.status.idle":"2022-07-12T12:44:38.885957Z","shell.execute_reply.started":"2022-07-12T12:43:12.566072Z","shell.execute_reply":"2022-07-12T12:44:38.883114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2p,ss,ssq,qt,qt2,rqt,rsp\n\ngmm = BayesianGaussianMixture(n_components = 8)\npreds = gmm.fit_predict(rsp)\noutput = pd.DataFrame({'Id': df.id, 'Predicted': preds})\noutput\noutput.to_csv('submission_rsp.csv', index=False)\nprint(\"Your submission was successfully saved!\")\nscoreGM = davies_bouldin_score(rsp, preds)\nscoreGM","metadata":{"execution":{"iopub.status.busy":"2022-07-12T12:48:49.403154Z","iopub.execute_input":"2022-07-12T12:48:49.403861Z","iopub.status.idle":"2022-07-12T12:50:03.235911Z","shell.execute_reply.started":"2022-07-12T12:48:49.403766Z","shell.execute_reply":"2022-07-12T12:50:03.233054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}