{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport matplotlib.pyplot as plt\nimport seaborn as sb\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T08:12:09.566476Z","iopub.execute_input":"2022-07-26T08:12:09.567491Z","iopub.status.idle":"2022-07-26T08:12:09.577212Z","shell.execute_reply.started":"2022-07-26T08:12:09.567441Z","shell.execute_reply":"2022-07-26T08:12:09.575724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **In the following Notebook we will looking at dataprep library and how it can help us**","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/tabular-playground-series-jul-2022/data.csv'\npath_sub = '/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:57:42.295834Z","iopub.execute_input":"2022-07-26T07:57:42.296320Z","iopub.status.idle":"2022-07-26T07:57:42.304617Z","shell.execute_reply.started":"2022-07-26T07:57:42.296275Z","shell.execute_reply":"2022-07-26T07:57:42.303481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(path)\nsub = pd.read_csv(path_sub)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:25:36.395586Z","iopub.execute_input":"2022-07-26T08:25:36.396181Z","iopub.status.idle":"2022-07-26T08:25:37.324625Z","shell.execute_reply.started":"2022-07-26T08:25:36.396132Z","shell.execute_reply":"2022-07-26T08:25:37.323352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Here we don't really need Id before the final submission hence I will dropping it for now**","metadata":{}},{"cell_type":"code","source":"df = df.drop(['id'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:57:43.733328Z","iopub.execute_input":"2022-07-26T07:57:43.733817Z","iopub.status.idle":"2022-07-26T07:57:43.755151Z","shell.execute_reply.started":"2022-07-26T07:57:43.733755Z","shell.execute_reply":"2022-07-26T07:57:43.753782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dataprep","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:57:43.756761Z","iopub.execute_input":"2022-07-26T07:57:43.758037Z","iopub.status.idle":"2022-07-26T07:58:35.204644Z","shell.execute_reply.started":"2022-07-26T07:57:43.757993Z","shell.execute_reply":"2022-07-26T07:58:35.203464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from dataprep.eda import plot, plot_correlation, create_report, plot_missing","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:58:35.205972Z","iopub.execute_input":"2022-07-26T07:58:35.206318Z","iopub.status.idle":"2022-07-26T07:58:37.655269Z","shell.execute_reply.started":"2022-07-26T07:58:35.206286Z","shell.execute_reply":"2022-07-26T07:58:37.654451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The following code will implement all basic plots of the dataset and will show a statistical summary of the data***","metadata":{}},{"cell_type":"code","source":"plot(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:58:37.656490Z","iopub.execute_input":"2022-07-26T07:58:37.656988Z","iopub.status.idle":"2022-07-26T07:58:47.565654Z","shell.execute_reply.started":"2022-07-26T07:58:37.656958Z","shell.execute_reply":"2022-07-26T07:58:47.564307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now lets look at further details such as correlation plots and feature by feature summary, which can be done by a simple command!**","metadata":{}},{"cell_type":"code","source":"create_report(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:58:47.567582Z","iopub.execute_input":"2022-07-26T07:58:47.567947Z","iopub.status.idle":"2022-07-26T07:59:22.782923Z","shell.execute_reply.started":"2022-07-26T07:58:47.567913Z","shell.execute_reply":"2022-07-26T07:59:22.781345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**This provides a great summary and a good place to start your EDA, now that we know about the data lets do some preprocessing and then build our Model**","metadata":{}},{"cell_type":"markdown","source":"**First from our EDA we have seen that some of the columns have distribution skewed to the left, hence it is a good idea to use power transformer which will help out gaussian model!**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\npt = PowerTransformer()\ndf_x = pt.fit_transform(df)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:59:22.784642Z","iopub.execute_input":"2022-07-26T07:59:22.785010Z","iopub.status.idle":"2022-07-26T07:59:26.579674Z","shell.execute_reply.started":"2022-07-26T07:59:22.784979Z","shell.execute_reply":"2022-07-26T07:59:26.578781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_x = pd.DataFrame(df_x,columns = df.columns)\ndf_x","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:59:26.582353Z","iopub.execute_input":"2022-07-26T07:59:26.583404Z","iopub.status.idle":"2022-07-26T07:59:26.630542Z","shell.execute_reply.started":"2022-07-26T07:59:26.583364Z","shell.execute_reply":"2022-07-26T07:59:26.628970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now lets build our first model BayesianGaussianMixture, now I have been practising this competition for quite a while now and n_components = 7 seems to be the best one**","metadata":{}},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture\nmodel1 = BayesianGaussianMixture(n_components = 7,covariance_type = \"full\",random_state = 1)\npred = model1.fit_predict(df_x)\npred","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:59:26.632466Z","iopub.execute_input":"2022-07-26T07:59:26.633039Z","iopub.status.idle":"2022-07-26T08:00:39.107468Z","shell.execute_reply.started":"2022-07-26T07:59:26.632949Z","shell.execute_reply":"2022-07-26T08:00:39.106141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now lets look at the plots of this prediction for better understanding the results**","metadata":{}},{"cell_type":"code","source":"plt.style.use('ggplot')\nplt.figure(figsize=(15,6))\nfor i in range(model1.means_.shape[0]):\n    plt.scatter(np.arange(df_x.shape[1]), model1.means_[i])\nplt.xticks(ticks=np.arange(df_x.shape[1]), labels=df.columns)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:00:39.109469Z","iopub.execute_input":"2022-07-26T08:00:39.110226Z","iopub.status.idle":"2022-07-26T08:00:39.510287Z","shell.execute_reply.started":"2022-07-26T08:00:39.110176Z","shell.execute_reply":"2022-07-26T08:00:39.509000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**This shows us and is also very common in the discussion and public notebooks that only a few of the coloumns \nseparate the clusters properly, hence it is a better idea to just use them!**","metadata":{}},{"cell_type":"code","source":"imp_cols = ['f_07','f_08','f_09','f_10','f_11','f_12','f_13','f_22','f_23','f_24','f_25','f_26','f_27','f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:11:39.972444Z","iopub.execute_input":"2022-07-26T08:11:39.972875Z","iopub.status.idle":"2022-07-26T08:11:39.978497Z","shell.execute_reply.started":"2022-07-26T08:11:39.972841Z","shell.execute_reply":"2022-07-26T08:11:39.977252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final = df_x[imp_cols]\ndf_final","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:16:33.272688Z","iopub.execute_input":"2022-07-26T08:16:33.273088Z","iopub.status.idle":"2022-07-26T08:16:33.307545Z","shell.execute_reply.started":"2022-07-26T08:16:33.273057Z","shell.execute_reply":"2022-07-26T08:16:33.306691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 12\nx = np.array(df_final)\ny = np.array(sub['Id'])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:30:10.673752Z","iopub.execute_input":"2022-07-26T08:30:10.674228Z","iopub.status.idle":"2022-07-26T08:30:10.685806Z","shell.execute_reply.started":"2022-07-26T08:30:10.674192Z","shell.execute_reply":"2022-07-26T08:30:10.684537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(i)):\n    model1.fit(x,y)\n    pred = model1.predict(x)\n    y = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:32:04.076211Z","iopub.execute_input":"2022-07-26T08:32:04.076588Z","iopub.status.idle":"2022-07-26T08:32:44.220210Z","shell.execute_reply.started":"2022-07-26T08:32:04.076557Z","shell.execute_reply":"2022-07-26T08:32:44.219089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = sub\nsubmission['Predicted'] = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:34:18.740661Z","iopub.execute_input":"2022-07-26T08:34:18.741054Z","iopub.status.idle":"2022-07-26T08:34:18.747276Z","shell.execute_reply.started":"2022-07-26T08:34:18.741026Z","shell.execute_reply":"2022-07-26T08:34:18.745837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:35:20.455902Z","iopub.execute_input":"2022-07-26T08:35:20.456520Z","iopub.status.idle":"2022-07-26T08:35:20.614915Z","shell.execute_reply.started":"2022-07-26T08:35:20.456471Z","shell.execute_reply":"2022-07-26T08:35:20.613671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}