{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"padding:20px;color:white;margin:0;font-size:200%;text-align:center;display:fill;border-radius:5px;background-color:#38A6A5;overflow:hidden;font-weight:500\">TPS July 2022</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nimport umap\n\nimport matplotlib.pyplot as plt\nimport matplotlib.colors\nfrom matplotlib.ticker import MaxNLocator\n\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.figure_factory as ff\n\nfrom sklearn import preprocessing\nfrom sklearn.manifold import TSNE\nfrom sklearn import metrics\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import davies_bouldin_score\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler,PowerTransformer,OrdinalEncoder, QuantileTransformer\n\nfrom tqdm import tqdm\n\nimport warnings, gc, string, random\nwarnings.filterwarnings(\"ignore\")\n\nfrom yellowbrick.cluster import KElbowVisualizer\n\nPATH_DATA = '/kaggle/input/tabular-playground-series-jul-2022/data.csv'\nPATH_SUBMISSION = '/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv'\n\ntemp=dict(layout=go.Layout(font=dict(family=\"Franklin Gothic\", size=12), \n                           height=500, width=1000))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-09T09:00:09.374407Z","iopub.execute_input":"2022-07-09T09:00:09.374869Z","iopub.status.idle":"2022-07-09T09:00:09.385884Z","shell.execute_reply.started":"2022-07-09T09:00:09.374833Z","shell.execute_reply":"2022-07-09T09:00:09.384125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b><span style='color:#444444'>1 |</span><span style='color:#38A6A5'> Data Analysis</span></b>","metadata":{}},{"cell_type":"code","source":"class CLR:\n    map_1 = 'GnBu'\n    map_2 = 'winter'\n    lb_t = '#88CAC9'\n    lb_d = '#38A6A5'\n    or_t = '#EDD3B3'\n    or_d = '#E1B580'\n    gr_t = '#DBDBDB'","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:09.443402Z","iopub.execute_input":"2022-07-09T09:00:09.44381Z","iopub.status.idle":"2022-07-09T09:00:09.449117Z","shell.execute_reply.started":"2022-07-09T09:00:09.443766Z","shell.execute_reply":"2022-07-09T09:00:09.447945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(PATH_DATA, index_col='id')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:09.518005Z","iopub.execute_input":"2022-07-09T09:00:09.518661Z","iopub.status.idle":"2022-07-09T09:00:10.239384Z","shell.execute_reply.started":"2022-07-09T09:00:09.518614Z","shell.execute_reply":"2022-07-09T09:00:10.238462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Dataframe shapes:', data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:10.241631Z","iopub.execute_input":"2022-07-09T09:00:10.24241Z","iopub.status.idle":"2022-07-09T09:00:10.248506Z","shell.execute_reply.started":"2022-07-09T09:00:10.242367Z","shell.execute_reply":"2022-07-09T09:00:10.247193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:10.249972Z","iopub.execute_input":"2022-07-09T09:00:10.250306Z","iopub.status.idle":"2022-07-09T09:00:10.283012Z","shell.execute_reply.started":"2022-07-09T09:00:10.250277Z","shell.execute_reply":"2022-07-09T09:00:10.281631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 98000 training samples. The Dataframes have an id column and 29 features. There are 22 float features, 7 int features and one string feature. There are no missing values:","metadata":{}},{"cell_type":"code","source":"int_features = data.select_dtypes('int')\nsub_titles=['Feature {}'.format(i.split('_')[-1]) for i in int_features.columns]\n\nprint(sub_titles)\n\nfig = make_subplots(rows=4, cols=2, subplot_titles=sub_titles)\nrow = 0\nc=[1,2]*4\n\nfor i, col in enumerate(int_features.columns):\n    if i % 2 == 0:\n        row += 1\n        fig.update_yaxes(title='Count',row=row,col=c[i])\n    df = int_features[col].value_counts().rename('count').reset_index()\n    #print(df)\n    fig.add_trace(go.Bar(x=df['index'], y=df['count'],\n                         marker_color=CLR.lb_t, marker_line=dict(color=CLR.lb_d),\n                         name='State', showlegend=(True if i==0 else False)),row=row, col=c[i])\n    \nfig.update_layout(template=temp,title=\"Distributions of integer features\",\n                  legend=dict(orientation=\"h\",yanchor=\"bottom\",y=1.03,xanchor=\"right\",x=.95),\n                  barmode='group',height=1500,width=900)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:10.285771Z","iopub.execute_input":"2022-07-09T09:00:10.286126Z","iopub.status.idle":"2022-07-09T09:00:10.414075Z","shell.execute_reply.started":"2022-07-09T09:00:10.286096Z","shell.execute_reply":"2022-07-09T09:00:10.412674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_features = data.select_dtypes('float64')\nsub_titles=['Feature {}'.format(i.split('_')[-1]) for i in float_features.columns]\n\nprint(sub_titles)\n\nfig = make_subplots(rows=8, cols=3, subplot_titles=sub_titles)\nrow = 0\nc=[1,2,3]*8\n\nfor i, col in enumerate(float_features.columns):\n    if i % 3 == 0:\n        row += 1\n        #print(i)\n        fig.update_yaxes(title='probability',row=row, col=c[i])\n    df = float_features[col].rename('feature').reset_index()\n    #print(df)\n    fig.add_trace(go.Histogram(x=df['feature'],nbinsx=80, histnorm='probability',\n                         marker_color=CLR.or_t, marker_line=dict(color=CLR.or_d),\n                         name='State', showlegend=(True if i==0 else False)),row=row, col=c[i])\n    \nfig.update_layout(template=temp,title=\"Distributions of float features\",\n                  legend=dict(orientation=\"h\",yanchor=\"bottom\",y=1.03,xanchor=\"right\",x=.95),\n                  barmode='group',height=1500,width=900)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:10.416033Z","iopub.execute_input":"2022-07-09T09:00:10.416377Z","iopub.status.idle":"2022-07-09T09:00:11.980932Z","shell.execute_reply.started":"2022-07-09T09:00:10.416347Z","shell.execute_reply":"2022-07-09T09:00:11.979703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=data.iloc[:,:-1].corr().round(2)  \nmask=np.triu(np.ones_like(corr, dtype=bool))\nc_mask = np.where(~mask, corr, 100)\nc=[]\nfor i in c_mask.tolist()[1:]:\n    c.append([x for x in i if x != 100])\n    \ncor=c[::-1]\nx=corr.index.tolist()[:-1]\ny=corr.columns.tolist()[1:][::-1]\nfig=ff.create_annotated_heatmap(z=cor, x=x, y=y,\n                                hovertemplate='Correlation between %{x} and %{y}= %{z}',\n                                colorscale='Peach', reversescale=True, name='')\nfig.update_layout(template=temp, title='Correlations between Features',\n                  yaxis=dict(showgrid=False,autorange=\"reversed\"),\n                  xaxis=dict(showgrid=False), height=1000,width=1000)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:11.982477Z","iopub.execute_input":"2022-07-09T09:00:11.982852Z","iopub.status.idle":"2022-07-09T09:00:12.405787Z","shell.execute_reply.started":"2022-07-09T09:00:11.98282Z","shell.execute_reply":"2022-07-09T09:00:12.404662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b><span style='color:#444444'>2 |</span><span style='color:#38A6A5'> application of PCA</span></b>","metadata":{}},{"cell_type":"code","source":"def anal_data(data):\n    pca_reducer = PCA()\n    data_reduced = pca_reducer.fit_transform(data)\n    return data_reduced","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:12.407217Z","iopub.execute_input":"2022-07-09T09:00:12.407595Z","iopub.status.idle":"2022-07-09T09:00:12.414057Z","shell.execute_reply.started":"2022-07-09T09:00:12.407563Z","shell.execute_reply":"2022-07-09T09:00:12.412515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature = anal_data(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:12.415818Z","iopub.execute_input":"2022-07-09T09:00:12.416979Z","iopub.status.idle":"2022-07-09T09:00:12.649253Z","shell.execute_reply.started":"2022-07-09T09:00:12.416933Z","shell.execute_reply":"2022-07-09T09:00:12.647382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cfm = pd.DataFrame(feature, columns=[\"PC{}\".format(x + 1) for x in range(len(data.columns))])\ndata_cfm.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:12.651579Z","iopub.execute_input":"2022-07-09T09:00:12.653263Z","iopub.status.idle":"2022-07-09T09:00:12.897228Z","shell.execute_reply.started":"2022-07-09T09:00:12.653214Z","shell.execute_reply":"2022-07-09T09:00:12.896051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value = np.random.rand(len(data))\nplt.figure(figsize=(6, 6))\nplt.scatter(feature[:, 0], feature[:, 1], alpha=0.8, c=value, cmap=CLR.map_2)\nplt.grid()\nplt.xlabel(\"PC1\")\nplt.ylabel(\"PC2\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:12.899716Z","iopub.execute_input":"2022-07-09T09:00:12.901487Z","iopub.status.idle":"2022-07-09T09:00:14.934453Z","shell.execute_reply.started":"2022-07-09T09:00:12.901412Z","shell.execute_reply":"2022-07-09T09:00:14.932776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b><span style='color:#444444'>4 |</span><span style='color:#38A6A5'> Modelling and Prediction</span></b>","metadata":{}},{"cell_type":"code","source":"Elbow_M = KElbowVisualizer(KMeans(random_state=1), k=15)\nElbow_M.fit(df)\nElbow_M.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:14.936842Z","iopub.execute_input":"2022-07-09T09:00:14.937371Z","iopub.status.idle":"2022-07-09T09:00:47.13535Z","shell.execute_reply.started":"2022-07-09T09:00:14.937322Z","shell.execute_reply":"2022-07-09T09:00:47.134007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so, we set number of clusters at 7.","metadata":{}},{"cell_type":"code","source":"def define_models(models=dict()):\n    random_state=1\n    number = 7\n    models['GM'] = GaussianMixture(n_components=7,random_state=random_state)\n    models['BGM'] = BayesianGaussianMixture(n_components=7,covariance_type='full', random_state=1, n_init=10)\n                                                                               \n    return models\n\ndef define_scalers(scalers=dict()):\n    scalers['robust'] = RobustScaler()\n    scalers['standard'] = StandardScaler()\n    \n    return scalers\n\ndef define_transformers(transformers=dict()):\n    transformers['power'] = PowerTransformer()\n    transformers['quantile'] = QuantileTransformer()\n    \n    return transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:47.13687Z","iopub.execute_input":"2022-07-09T09:00:47.137212Z","iopub.status.idle":"2022-07-09T09:00:47.144918Z","shell.execute_reply.started":"2022-07-09T09:00:47.13718Z","shell.execute_reply":"2022-07-09T09:00:47.143847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_data(df, scaler, transformer):\n    X_scaled = scaler.fit(df).transform(df)\n    X_scaled = transformer.fit(X_scaled).transform(X_scaled)\n    X_scaled = pd.DataFrame(X_scaled, columns = df.columns)\n    return X_scaled\n\ndef evaluate_single_model(model, df, scaler, transformer):\n    data_scaled = scale_data(df, scaler, transformer)\n    pred = model.fit_predict(data_scaled)\n    proxy_score = davies_bouldin_score(df, pred)\n    return proxy_score, pred\n\ndef score_models(df, models, scalers, transformers):\n    results = dict()\n    outputs = dict()\n    for model_name, model in tqdm(models.items()):\n        for scaler_name, scaler in scalers.items():\n            for transformer_name, transformer in transformers.items():\n                    results[model_name, scaler_name, transformer_name], outputs[model_name, scaler_name, transformer_name] = evaluate_single_model(model, df, scaler, transformer)\n    return results, outputs\n\ndef score_summary(results):\n    scores = [(k,v) for k,v in results.items()]\n    scores = sorted(scores, key=lambda x: x[1])\n    for name, score in scores:\n        print(f'Score={score:.3f}\\t{name}')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:12:15.261198Z","iopub.execute_input":"2022-07-09T09:12:15.26163Z","iopub.status.idle":"2022-07-09T09:12:15.273367Z","shell.execute_reply.started":"2022-07-09T09:12:15.261599Z","shell.execute_reply":"2022-07-09T09:12:15.272172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results, outputs = score_models(data,define_models(),define_scalers(),define_transformers())","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:12:15.611838Z","iopub.execute_input":"2022-07-09T09:12:15.612813Z","iopub.status.idle":"2022-07-09T10:04:53.716433Z","shell.execute_reply.started":"2022-07-09T09:12:15.612773Z","shell.execute_reply":"2022-07-09T10:04:53.715176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_summary(results)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:04:53.71897Z","iopub.execute_input":"2022-07-09T10:04:53.71984Z","iopub.status.idle":"2022-07-09T10:04:53.736305Z","shell.execute_reply.started":"2022-07-09T10:04:53.71979Z","shell.execute_reply":"2022-07-09T10:04:53.73382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <b><span style='color:#444444'>3 |</span><span style='color:#38A6A5'>submission</span></b>","metadata":{}},{"cell_type":"code","source":"def make_submission(outputs):\n    for key in outputs:\n        submission = pd.read_csv(PATH_SUBMISSION)\n        title = key[0]+'_'+key[1]+'_'+key[2]    \n        submission['Predicted'] = outputs[key]\n        submission.head()\n        pl = sns.countplot(x=submission[\"Predicted\"])\n        pl.set_title(f'Distribution Of {title}')\n        plt.show()\n        submission.to_csv(f'{title}.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:04:53.745336Z","iopub.execute_input":"2022-07-09T10:04:53.746567Z","iopub.status.idle":"2022-07-09T10:04:53.756165Z","shell.execute_reply.started":"2022-07-09T10:04:53.746511Z","shell.execute_reply":"2022-07-09T10:04:53.754764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"make_submission(outputs)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:04:53.759636Z","iopub.execute_input":"2022-07-09T10:04:53.760925Z","iopub.status.idle":"2022-07-09T10:04:56.330106Z","shell.execute_reply.started":"2022-07-09T10:04:53.760879Z","shell.execute_reply":"2022-07-09T10:04:56.328843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}