{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import PowerTransformer,MinMaxScaler,RobustScaler,PolynomialFeatures\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans, DBSCAN\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.metrics import silhouette_score\nfrom tqdm import tqdm\nimport keras\nfrom keras import layers\nimport tensorflow as tf\n\nimport matplotlib.pyplot as plt  # for plotting","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T14:39:54.307106Z","iopub.execute_input":"2022-07-22T14:39:54.307732Z","iopub.status.idle":"2022-07-22T14:40:01.247284Z","shell.execute_reply.started":"2022-07-22T14:39:54.307606Z","shell.execute_reply":"2022-07-22T14:40:01.245801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\")\ntrain_all = train.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:01.250066Z","iopub.execute_input":"2022-07-22T14:40:01.251039Z","iopub.status.idle":"2022-07-22T14:40:02.779806Z","shell.execute_reply.started":"2022-07-22T14:40:01.250980Z","shell.execute_reply":"2022-07-22T14:40:02.778442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:02.781557Z","iopub.execute_input":"2022-07-22T14:40:02.782110Z","iopub.status.idle":"2022-07-22T14:40:02.792791Z","shell.execute_reply.started":"2022-07-22T14:40:02.782056Z","shell.execute_reply":"2022-07-22T14:40:02.791478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_all.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:02.797184Z","iopub.execute_input":"2022-07-22T14:40:02.797814Z","iopub.status.idle":"2022-07-22T14:40:02.842410Z","shell.execute_reply.started":"2022-07-22T14:40:02.797745Z","shell.execute_reply":"2022-07-22T14:40:02.841062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:02.844169Z","iopub.execute_input":"2022-07-22T14:40:02.844610Z","iopub.status.idle":"2022-07-22T14:40:02.879453Z","shell.execute_reply.started":"2022-07-22T14:40:02.844574Z","shell.execute_reply":"2022-07-22T14:40:02.878159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['id'], axis = 1, inplace = True)\npoly = PolynomialFeatures(include_bias = False)\ntrain = poly.fit_transform(train)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:02.881963Z","iopub.execute_input":"2022-07-22T14:40:02.882763Z","iopub.status.idle":"2022-07-22T14:40:03.430965Z","shell.execute_reply.started":"2022-07-22T14:40:02.882708Z","shell.execute_reply":"2022-07-22T14:40:03.430026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# check multicollinearity\ndef corrFilter(x: pd.DataFrame, bound: float):\n    xCorr = x.corr()\n    xFiltered = xCorr[((xCorr >= bound) | (xCorr <= -bound)) & (xCorr !=1.000)]\n    xFlattened = xFiltered.unstack().sort_values().drop_duplicates()\n    return xFlattened\n\ncorrFilter(pd.DataFrame(train), .98)\n\n# there is no multicollinearity problem, you can change it to code and run.","metadata":{"execution":{"iopub.status.busy":"2022-07-20T20:43:26.348976Z","iopub.execute_input":"2022-07-20T20:43:26.349385Z","iopub.status.idle":"2022-07-20T20:44:15.189005Z","shell.execute_reply.started":"2022-07-20T20:43:26.349348Z","shell.execute_reply":"2022-07-20T20:44:15.187009Z"}}},{"cell_type":"code","source":"pca = PCA(n_components=10)\npca.fit(train)\nprint(pca.explained_variance_ratio_)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:03.432386Z","iopub.execute_input":"2022-07-22T14:40:03.433339Z","iopub.status.idle":"2022-07-22T14:40:08.199953Z","shell.execute_reply.started":"2022-07-22T14:40:03.433244Z","shell.execute_reply":"2022-07-22T14:40:08.198275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components=7)\npca_cols = pca.fit_transform(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:08.201785Z","iopub.execute_input":"2022-07-22T14:40:08.202310Z","iopub.status.idle":"2022-07-22T14:40:12.285828Z","shell.execute_reply.started":"2022-07-22T14:40:08.202256Z","shell.execute_reply":"2022-07-22T14:40:12.284499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:12.287232Z","iopub.execute_input":"2022-07-22T14:40:12.287634Z","iopub.status.idle":"2022-07-22T14:40:12.297714Z","shell.execute_reply.started":"2022-07-22T14:40:12.287602Z","shell.execute_reply":"2022-07-22T14:40:12.295776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components=10)\npca.fit(train_all)\nprint(pca.explained_variance_ratio_)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:12.301853Z","iopub.execute_input":"2022-07-22T14:40:12.302995Z","iopub.status.idle":"2022-07-22T14:40:13.117218Z","shell.execute_reply.started":"2022-07-22T14:40:12.302943Z","shell.execute_reply":"2022-07-22T14:40:13.115093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components=1)\npca_col = pca.fit_transform(train_all)\ntrain_all['pca'] = pca_col","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:13.119673Z","iopub.execute_input":"2022-07-22T14:40:13.123094Z","iopub.status.idle":"2022-07-22T14:40:13.948505Z","shell.execute_reply.started":"2022-07-22T14:40:13.123004Z","shell.execute_reply":"2022-07-22T14:40:13.946660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols=[col for col in train_all.columns]\n# idea for selected features comes from https://www.kaggle.com/competitions/tabular-playground-series-jul-2022/discussion/334875, also add the pca column\nbest_cols =['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28','pca']\ndrop_cols=[col for col in all_cols if col not in best_cols]\ntrain_all=train_all.drop(drop_cols,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:13.956808Z","iopub.execute_input":"2022-07-22T14:40:13.957994Z","iopub.status.idle":"2022-07-22T14:40:13.989797Z","shell.execute_reply.started":"2022-07-22T14:40:13.957918Z","shell.execute_reply":"2022-07-22T14:40:13.988589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([pd.DataFrame(train_all), pd.DataFrame(pca_cols)], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:13.992393Z","iopub.execute_input":"2022-07-22T14:40:13.992979Z","iopub.status.idle":"2022-07-22T14:40:14.012878Z","shell.execute_reply.started":"2022-07-22T14:40:13.992920Z","shell.execute_reply":"2022-07-22T14:40:14.010496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:14.014437Z","iopub.execute_input":"2022-07-22T14:40:14.015941Z","iopub.status.idle":"2022-07-22T14:40:14.047637Z","shell.execute_reply.started":"2022-07-22T14:40:14.015841Z","shell.execute_reply":"2022-07-22T14:40:14.046419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns = [str(i) for i in train.columns.tolist()]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:14.050212Z","iopub.execute_input":"2022-07-22T14:40:14.050800Z","iopub.status.idle":"2022-07-22T14:40:14.057556Z","shell.execute_reply.started":"2022-07-22T14:40:14.050729Z","shell.execute_reply":"2022-07-22T14:40:14.056402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"robust = RobustScaler(); power = PowerTransformer()\ntrain = power.fit_transform(train)\ntrain = robust.fit_transform(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:14.059084Z","iopub.execute_input":"2022-07-22T14:40:14.060225Z","iopub.status.idle":"2022-07-22T14:40:17.390152Z","shell.execute_reply.started":"2022-07-22T14:40:14.060179Z","shell.execute_reply":"2022-07-22T14:40:17.388545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(train).describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.391851Z","iopub.execute_input":"2022-07-22T14:40:17.392265Z","iopub.status.idle":"2022-07-22T14:40:17.598674Z","shell.execute_reply.started":"2022-07-22T14:40:17.392230Z","shell.execute_reply":"2022-07-22T14:40:17.596708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hyper pruning parameters","metadata":{}},{"cell_type":"markdown","source":"for i in tqdm(range(2,15+1)):\n    kmeans = KMeans(n_clusters=i, random_state=1234).fit(train)\n    silhouette = silhouette_score(train,\n            kmeans.labels_,\n            metric=\"euclidean\",sample_size = 300, random_state = 1234\n        )\n    print(\"k = {}, silhouette score:{}\".format(i,silhouette))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:07:29.974629Z","iopub.execute_input":"2022-07-22T14:07:29.975030Z","iopub.status.idle":"2022-07-22T14:08:31.814596Z","shell.execute_reply.started":"2022-07-22T14:07:29.974996Z","shell.execute_reply":"2022-07-22T14:08:31.810574Z"}}},{"cell_type":"markdown","source":"for i in tqdm(range(2,15+1)):\n    bayes = BayesianGaussianMixture(n_components=i, random_state=1234).fit(train)\n    silhouette = silhouette_score(train,\n            bayes.predict(train),\n            metric=\"euclidean\",sample_size = 300, random_state = 1234\n        )\n    print(\"k = {}, silhouette score:{}\".format(i,silhouette))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:16:47.163370Z","iopub.execute_input":"2022-07-22T14:16:47.163776Z","iopub.status.idle":"2022-07-22T14:21:57.199699Z","shell.execute_reply.started":"2022-07-22T14:16:47.163741Z","shell.execute_reply":"2022-07-22T14:21:57.193182Z"}}},{"cell_type":"code","source":"model = KMeans(n_clusters=5, random_state=1234).fit(train)\n#model = BayesianGaussianMixture(n_components=5, covariance_type='full', max_iter=200, random_state=1234).fit(train)\n#model = DBSCAN(eps=4, min_samples=3).fit(train)\nsubmission=pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\n#submission['Predicted']=model.predict(train)\nsubmission['Predicted']=model.labels_\nsubmission.to_csv(\"submission.csv\",index=False)\nsubmission.head()# score not high, 19.3, it's a try on leader board","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:45:21.450020Z","iopub.execute_input":"2022-07-22T14:45:21.450507Z","iopub.status.idle":"2022-07-22T14:45:25.907574Z","shell.execute_reply.started":"2022-07-22T14:45:21.450468Z","shell.execute_reply":"2022-07-22T14:45:25.906314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raise SystemExit()\n# code below uses auto encoder approach, may take a longer time to run","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.948148Z","iopub.status.idle":"2022-07-22T14:40:17.948930Z","shell.execute_reply.started":"2022-07-22T14:40:17.948674Z","shell.execute_reply":"2022-07-22T14:40:17.948697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The auto encoder’s performance is not high enough to use this. The main reason is number of attributes not high enough. So this method we can choose to apply if there is a high number of attributes.","metadata":{}},{"cell_type":"code","source":"shape = train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.950073Z","iopub.status.idle":"2022-07-22T14:40:17.950489Z","shell.execute_reply.started":"2022-07-22T14:40:17.950293Z","shell.execute_reply":"2022-07-22T14:40:17.950312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.reshape(-1,shape[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.952051Z","iopub.status.idle":"2022-07-22T14:40:17.952523Z","shell.execute_reply.started":"2022-07-22T14:40:17.952291Z","shell.execute_reply":"2022-07-22T14:40:17.952311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder_input_data = keras.Input(shape=(shape[1]))\nencoded_layer1 = keras.layers.Dense(16, activation='relu')(encoder_input_data)\nencoded_layer2 = keras.layers.Dense(8, activation='relu')(encoded_layer1)\ndecoded_layer1 = keras.layers.Dense(16, activation='relu')(encoded_layer2)\ndecoded_layer2 = keras.layers.Dense(shape[1], activation='sigmoid')(decoded_layer1)\nencoder_model = keras.Model(encoder_input_data, decoded_layer2)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.954128Z","iopub.status.idle":"2022-07-22T14:40:17.954548Z","shell.execute_reply.started":"2022-07-22T14:40:17.954353Z","shell.execute_reply":"2022-07-22T14:40:17.954371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder_model.compile(optimizer=\"RMSprop\", metrics = [\"accuracy\"], loss=tf.keras.losses.mean_squared_error)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.955960Z","iopub.status.idle":"2022-07-22T14:40:17.956365Z","shell.execute_reply.started":"2022-07-22T14:40:17.956168Z","shell.execute_reply":"2022-07-22T14:40:17.956185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.959174Z","iopub.status.idle":"2022-07-22T14:40:17.960219Z","shell.execute_reply.started":"2022-07-22T14:40:17.959719Z","shell.execute_reply":"2022-07-22T14:40:17.959753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# auto encoder structure\ntf.keras.utils.plot_model(model=encoder_model, rankdir=\"LR\", dpi=130, show_shapes=True, to_file=\"autoencoder.png\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.962169Z","iopub.status.idle":"2022-07-22T14:40:17.962701Z","shell.execute_reply.started":"2022-07-22T14:40:17.962488Z","shell.execute_reply":"2022-07-22T14:40:17.962508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, test_data, train_labels, test_labels = train_test_split(train,np.zeros(len(train)), train_size=0.8, random_state=1234)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.964066Z","iopub.status.idle":"2022-07-22T14:40:17.964583Z","shell.execute_reply.started":"2022-07-22T14:40:17.964366Z","shell.execute_reply":"2022-07-22T14:40:17.964386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = encoder_model.fit(train_data,train_data,epochs = 100,batch_size=256, shuffle=True, validation_data = (test_data, test_data))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.967207Z","iopub.status.idle":"2022-07-22T14:40:17.967697Z","shell.execute_reply.started":"2022-07-22T14:40:17.967480Z","shell.execute_reply":"2022-07-22T14:40:17.967500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_fig, (ax1, ax2) = plt.subplots(2, sharex=True)\nhistory_fig.suptitle('Autoencoder Training Performance')\nax1.plot(range(0,100), history.history['accuracy'], color='blue')\nax1.set(ylabel='Reconstruction Accuracy')\nax2.plot(range(0,100), np.log10(history.history['loss']), color='blue')\nax2.plot(range(0,100), np.log10(history.history['val_loss']), color='red', alpha=0.9)\nax2.set(ylabel='log_10(loss)', xlabel='Training Epoch')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.969963Z","iopub.status.idle":"2022-07-22T14:40:17.970647Z","shell.execute_reply.started":"2022-07-22T14:40:17.970399Z","shell.execute_reply":"2022-07-22T14:40:17.970426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = encoder_model.predict(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.971926Z","iopub.status.idle":"2022-07-22T14:40:17.972513Z","shell.execute_reply.started":"2022-07-22T14:40:17.972292Z","shell.execute_reply":"2022-07-22T14:40:17.972316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.975115Z","iopub.status.idle":"2022-07-22T14:40:17.975756Z","shell.execute_reply.started":"2022-07-22T14:40:17.975479Z","shell.execute_reply":"2022-07-22T14:40:17.975502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(2,15+1)):\n    kmeans = KMeans(n_clusters=i, random_state=1234).fit(train)\n    silhouette = silhouette_score(train,\n            kmeans.labels_,\n            metric=\"euclidean\",sample_size = 1000, random_state = 1234\n        )\n    print(\"k = {}, silhouette score:{}\".format(i,silhouette))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.978020Z","iopub.status.idle":"2022-07-22T14:40:17.978533Z","shell.execute_reply.started":"2022-07-22T14:40:17.978312Z","shell.execute_reply":"2022-07-22T14:40:17.978332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(range(2,15+1)):\n    bayes = BayesianGaussianMixture(n_components=i, covariance_type='full', max_iter=200, random_state=1234).fit(train)\n    silhouette = silhouette_score(train,\n            bayes.predict(train),\n            metric=\"euclidean\",sample_size = 1000, random_state = 1234\n        )\n    print(\"k = {}, silhouette score:{}\".format(i,silhouette))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.980311Z","iopub.status.idle":"2022-07-22T14:40:17.981016Z","shell.execute_reply.started":"2022-07-22T14:40:17.980763Z","shell.execute_reply":"2022-07-22T14:40:17.980797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kmeans: silhouette score choose k = 8, 0.13230429589748383","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(n_clusters=8, random_state=1234).fit(train)\nkmeans.labels_\nsubmission=pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsubmission['Predicted']=kmeans.labels_\nsubmission.to_csv(\"submission.csv\",index=False)\nsubmission.head()# score not high, 19.3, it's a try on leader board","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:17.982047Z","iopub.status.idle":"2022-07-22T14:40:17.982459Z","shell.execute_reply.started":"2022-07-22T14:40:17.982261Z","shell.execute_reply":"2022-07-22T14:40:17.982280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reference: https://matthew-parker.rbind.io/post/2021-01-16-pytorch-keras-clustering/\nhttps://blog.keras.io/building-autoencoders-in-keras.html\nhttps://www.kaggle.com/code/taos2000/mnist-cnn-vgg16-lenet-5 (my other keras notebook)","metadata":{}}]}