{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. EDA\n- Onehot encoding ワンホットエンコーディング\n- Scaling スケーリング\n- Zero filling ゼロ埋め","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T18:16:29.205092Z","iopub.execute_input":"2022-08-12T18:16:29.205548Z","iopub.status.idle":"2022-08-12T18:16:29.211301Z","shell.execute_reply.started":"2022-08-12T18:16:29.205506Z","shell.execute_reply":"2022-08-12T18:16:29.210003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample size サンプル件数\nn=10000\ndf_train = pd.read_feather('../input/amexfeather/train_data.ftr').sample(n=n)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T18:56:33.215800Z","iopub.execute_input":"2022-08-12T18:56:33.217341Z","iopub.status.idle":"2022-08-12T18:56:33.222695Z","shell.execute_reply.started":"2022-08-12T18:56:33.217268Z","shell.execute_reply":"2022-08-12T18:56:33.221607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Excluding ID and date information. IDと日付情報を削除。\ndf_train = df_train.drop(['customer_ID','S_2'], axis=1)\n\n# Onehot encoding. カテゴリ変数をOnehotエンコーディング。\ndf_train = pd.get_dummies(df_train)\n\n# Explanatory Variable Extraction. 最後の列targetを除いて説明変数とする。\nx = df_train.iloc[:, :-1]\n\n# Scaling. オートスケーリング。\nautoscaled_x = (x - x.mean()) / x.std()\n\n# Zero filling. 欠損値0埋め。\nautoscaled_x = autoscaled_x.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:17:18.757988Z","iopub.execute_input":"2022-08-12T19:17:18.758805Z","iopub.status.idle":"2022-08-12T19:17:18.907917Z","shell.execute_reply.started":"2022-08-12T19:17:18.758760Z","shell.execute_reply":"2022-08-12T19:17:18.906980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Dimension Reduction 次元削減\n### 2-1. PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:17:52.780228Z","iopub.execute_input":"2022-08-12T19:17:52.780634Z","iopub.status.idle":"2022-08-12T19:17:52.785928Z","shell.execute_reply.started":"2022-08-12T19:17:52.780598Z","shell.execute_reply":"2022-08-12T19:17:52.784812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components=2)\nvecs_list_pca = pca.fit_transform(autoscaled_x)\nprint(pca.explained_variance_ratio_)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:17:54.086217Z","iopub.execute_input":"2022-08-12T19:17:54.087267Z","iopub.status.idle":"2022-08-12T19:17:54.456326Z","shell.execute_reply.started":"2022-08-12T19:17:54.087225Z","shell.execute_reply":"2022-08-12T19:17:54.454906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2-2. SVD","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import TruncatedSVD","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:17:59.192584Z","iopub.execute_input":"2022-08-12T19:17:59.192995Z","iopub.status.idle":"2022-08-12T19:17:59.199356Z","shell.execute_reply.started":"2022-08-12T19:17:59.192960Z","shell.execute_reply":"2022-08-12T19:17:59.197921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svd = TruncatedSVD(n_components=2)\nvecs_list_svd = svd.fit_transform(autoscaled_x)\nprint(svd.explained_variance_ratio_)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:18:05.346642Z","iopub.execute_input":"2022-08-12T19:18:05.347073Z","iopub.status.idle":"2022-08-12T19:18:05.617187Z","shell.execute_reply.started":"2022-08-12T19:18:05.347037Z","shell.execute_reply":"2022-08-12T19:18:05.615622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2-3. t-SNE","metadata":{}},{"cell_type":"code","source":"from sklearn.manifold import TSNE","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:18:16.497597Z","iopub.execute_input":"2022-08-12T19:18:16.497999Z","iopub.status.idle":"2022-08-12T19:18:16.503315Z","shell.execute_reply.started":"2022-08-12T19:18:16.497963Z","shell.execute_reply":"2022-08-12T19:18:16.501982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsne = TSNE(n_components=2,\n            perplexity=30,\n            n_iter=1000\n           )\nvecs_list_tsne = tsne.fit_transform(autoscaled_x)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:40:45.821659Z","iopub.execute_input":"2022-08-12T19:40:45.822083Z","iopub.status.idle":"2022-08-12T19:41:46.744932Z","shell.execute_reply.started":"2022-08-12T19:40:45.822048Z","shell.execute_reply":"2022-08-12T19:41:46.743880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2-4. UMAP","metadata":{}},{"cell_type":"code","source":"import umap.umap_ as umap\nfrom scipy.sparse.csgraph import connected_components","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:26:30.743893Z","iopub.execute_input":"2022-08-12T19:26:30.744714Z","iopub.status.idle":"2022-08-12T19:26:30.767181Z","shell.execute_reply.started":"2022-08-12T19:26:30.744670Z","shell.execute_reply":"2022-08-12T19:26:30.765905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_umap = umap.UMAP(n_components=2,\n                       n_neighbors=5,\n                       min_dist=0.1,\n                       metric='euclidean'\n                      )\nvecs_list_umap = model_umap.fit_transform(autoscaled_x)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:32:44.329099Z","iopub.execute_input":"2022-08-12T19:32:44.329700Z","iopub.status.idle":"2022-08-12T19:32:54.327173Z","shell.execute_reply.started":"2022-08-12T19:32:44.329648Z","shell.execute_reply":"2022-08-12T19:32:54.325661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Visualization 可視化","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(15, 15))\n\nsns.scatterplot(x=vecs_list_pca[:,0],y=vecs_list_pca[:,1],\n                hue=df_train['target'],alpha=0.5,ax = axes[0,0])\naxes[0,0].set_title('PCA')\n\nsns.scatterplot(x=vecs_list_svd[:,0],y=vecs_list_svd[:,1],\n                hue=df_train['target'],alpha=0.5,ax = axes[0,1])\naxes[0,1].set_title('SVD')\n\nsns.scatterplot(x=vecs_list_tsne[:,0],y=vecs_list_tsne[:,1],\n                hue=df_train['target'],alpha=0.5,ax = axes[1,0])\naxes[1,0].set_title('t-SNE')\n\nsns.scatterplot(x=vecs_list_umap[:,0],y=vecs_list_umap[:,1],\n                hue=df_train['target'],alpha=0.5,ax = axes[1,1])\naxes[1,1].set_title('UMAP')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:41:46.746983Z","iopub.execute_input":"2022-08-12T19:41:46.747747Z","iopub.status.idle":"2022-08-12T19:41:49.683406Z","shell.execute_reply.started":"2022-08-12T19:41:46.747705Z","shell.execute_reply":"2022-08-12T19:41:49.682228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- There is clusters of only payers.<br>\n支払う人だけのクラスタが存在する。\n- There is clusters with a mixture of payers and non-payers.<br>\n支払う人、しない人が混在するクラスタが存在する。\n- There is no cluster with only non-payers, but there is clusters with a very large number of non-payers.<br>\n支払わない人だけのクラスタは存在しないが、支払わない人が非常に多いクラスタは存在する。","metadata":{}}]}