{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T10:27:08.438103Z","iopub.execute_input":"2022-08-02T10:27:08.439012Z","iopub.status.idle":"2022-08-02T10:27:08.451014Z","shell.execute_reply.started":"2022-08-02T10:27:08.438960Z","shell.execute_reply":"2022-08-02T10:27:08.449488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Previous result: \n1. EDA(Missing value)\nhttps://www.kaggle.com/code/makotouchiyama/english-eda-1-missingval-r-tps-202208","metadata":{}},{"cell_type":"markdown","source":"# Purpose of this analysis\n\nEDA for pre-feature selection in measurement1~17\n\n ※Missing values are imputed by SimpleImputer(strategy='most_frequent')\n \n1. Checking cluster_number in each features roughly to exert two\n    ⇒ Unrecognizable in this analysis.\n2. Mann-whitney U-test between failure and normal condition in each features\n\nConclusion\n\nStatistical analysis weakly suggests that several features have the potential to be useful features in modeling using decision tree type algothythm.   \n\nseveral features (P val < 0.05)(\"measurement_2\",\"measurement_6\",\"measurement_7\",\"measurement_8\")\n\n===========================================================================================\n\n\nmeasurement 0~17における特徴量選択前解析\n　　\n  ※欠損値は最頻値を用いて補完\n\n解析の流れ\n\n① data中（measurement0~17）に認識できないクラスターが存在するか確認した。\n\n    ⇒確認できなかったため、２郡での比較を実施し、有意差のある特徴量の抽出を試みた。\n\n② Mann-whitney U-test（ノンパラメトリック検定）を用いて、有意に異なる特徴量を抽出した\n        \n        (threshold P val <0.05)。\n        \n結論\n\n統計的手法によりエビデンスとしては弱いが有用な特徴量となりうる可能性を秘めた特徴量を抽出できた。\n決定着タイプのアルゴリズムには有用かもしれない\n(\"measurement_2\",\"measurement_6\",\"measurement_7\",\"measurement_8\")\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import RobustScaler, PowerTransformer\nfrom sklearn import preprocessing\nfrom sklearn.cluster import KMeans\nfrom matplotlib import cm\nfrom sklearn import metrics\nfrom sklearn.impute import SimpleImputer\nfrom scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:27:08.453717Z","iopub.execute_input":"2022-08-02T10:27:08.454534Z","iopub.status.idle":"2022-08-02T10:27:08.467020Z","shell.execute_reply.started":"2022-08-02T10:27:08.454482Z","shell.execute_reply":"2022-08-02T10:27:08.465670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import data、scaling data\ndata_path = '../input/tabular-playground-series-aug-2022/train.csv'\ndf_train = pd.read_csv(data_path)\n\nimpute_data = pd.DataFrame(SimpleImputer(strategy='most_frequent').fit_transform(df_train),columns=df_train.columns)\nclustering_data = impute_data.iloc[:,7:impute_data.shape[1]]\n\nclustering_data_all = clustering_data.drop(columns='failure')\nscaled_data_all=preprocessing.RobustScaler().fit_transform(clustering_data_all)\nscaled_data_all=preprocessing.PowerTransformer().fit_transform(scaled_data_all)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:27:08.490986Z","iopub.execute_input":"2022-08-02T10:27:08.491563Z","iopub.status.idle":"2022-08-02T10:27:09.658621Z","shell.execute_reply.started":"2022-08-02T10:27:08.491516Z","shell.execute_reply":"2022-08-02T10:27:09.657359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_data = scaled_data_all\n\nscore=[]\nsum_of_squared_errors = []\nfor i in range(2, 15):\n    model = KMeans(n_clusters=i, random_state=0, init='random')\n    model.fit(scaled_data)\n    sum_of_squared_errors.append(model.inertia_)\n    prediction=model.predict(scaled_data)\n    \n    shscore=metrics.silhouette_score(scaled_data,prediction, metric='euclidean')\n    chscore=metrics.calinski_harabasz_score(scaled_data,prediction)\n    dbscore=metrics.davies_bouldin_score(scaled_data,prediction)\n    score.append([i,shscore,chscore,dbscore])\n\nplt.plot(range(2, 15), sum_of_squared_errors, marker='o')\nplt.xlabel('number of clusters')\nplt.ylabel('sum of squared errors')\nplt.show()\n\nscore=pd.DataFrame(score)\nscore.columns = ['cluster_num','silhouette','calinski_harabasz','davies_bouldin'] \n\nprint(pd.DataFrame(score))\n\nplt.subplot(3,1,1)\nplt.plot(score.iloc[:,0],score.iloc[:,1],\n        color=cm.Set1.colors[1], label=\"silhouette_score\")\nplt.legend() \nplt.xticks(color='w')\n\nplt.subplot(3,1,2)\nplt.plot(score.iloc[:,0],score.iloc[:,2],\n        color=cm.Set1.colors[0], label=\"calinski_harabasz_score\")\nplt.legend()\nplt.xticks(color='w')\n\nplt.subplot(3,1,3)\nplt.plot(score.iloc[:,0],score.iloc[:,3],\n        color=cm.Set1.colors[2], label=\"davies_bouldin_score\")\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:27:09.662421Z","iopub.execute_input":"2022-08-02T10:27:09.662970Z","iopub.status.idle":"2022-08-02T10:29:40.708396Z","shell.execute_reply.started":"2022-08-02T10:27:09.662919Z","shell.execute_reply":"2022-08-02T10:29:40.707087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mann-Whitney U-test\nMannWhitneyResult=[]\n\nscaled_data_all = pd.DataFrame(scaled_data_all,columns=clustering_data.drop(columns='failure').columns)\nscaled_data_all_label=pd.concat([scaled_data_all,clustering_data['failure']],axis=1)\n\nnormal = scaled_data_all_label[scaled_data_all_label['failure'] == 0]\nfailure = scaled_data_all_label[scaled_data_all_label['failure'] == 1]\n\n# down sampling for statistical analysis (sample_num(normal) > sample_num(failure))\nnormal = normal.sample(n=failure.shape[0], random_state=1)\n\n\nfor i in range(0,17):\n    normal_feature=normal.iloc[:,i]\n    failure_feature=failure.iloc[:,i]\n    result=stats.mannwhitneyu(normal_feature, failure_feature, alternative='two-sided')\n    inverse_pval=1/result[1]\n    normal_feature_median = normal_feature.median()\n    failure_feature_median = failure_feature.median()\n    delta_med=normal_feature_median-failure_feature_median \n    MannWhitneyResult.append([scaled_data_all_label.columns[i],result[1],inverse_pval,delta_med,normal_feature_median,failure_feature_median])\n\nMannWhitneyResult=pd.DataFrame(MannWhitneyResult)\nMannWhitneyResult.columns = ['feature','Pval','1/Pval','delta(median)','median(normal)','median(normal)']\nprint(MannWhitneyResult)\n\nplt.bar(x=MannWhitneyResult['feature'],height=MannWhitneyResult['1/Pval'])\nplt.xticks(rotation=90)\nplt.ylabel('1/Pval')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:29:40.710057Z","iopub.execute_input":"2022-08-02T10:29:40.710425Z","iopub.status.idle":"2022-08-02T10:29:41.180233Z","shell.execute_reply.started":"2022-08-02T10:29:40.710390Z","shell.execute_reply":"2022-08-02T10:29:41.178901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"useful_cols = [\"measurement_2\",\"measurement_6\",\"measurement_7\",\"measurement_8\"]\nuseful_cols_label=useful_cols\nuseful_cols_label.append(\"failure\")\nscaled_data_all_label_uf = scaled_data_all_label[useful_cols_label]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:29:41.183093Z","iopub.execute_input":"2022-08-02T10:29:41.183440Z","iopub.status.idle":"2022-08-02T10:29:41.191276Z","shell.execute_reply.started":"2022-08-02T10:29:41.183408Z","shell.execute_reply":"2022-08-02T10:29:41.189772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# histgraghm of candidate of useful features\n\nnormal_df=normal\nfailure_df=failure\n\nplt.subplots(figsize=(25,35))\nnormal_df[\"df\"] = \"normal\"\nfailure_df[\"df\"] = \"failure\"\nfor i, column in enumerate(useful_cols):\n    plt.subplot(6,3,i+1)\n    sns.histplot(data=pd.concat([normal_df, failure_df]).reset_index(drop=True), x=column,hue=\"df\")\n    plt.title(column)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:29:41.192905Z","iopub.execute_input":"2022-08-02T10:29:41.193236Z","iopub.status.idle":"2022-08-02T10:29:43.411992Z","shell.execute_reply.started":"2022-08-02T10:29:41.193207Z","shell.execute_reply":"2022-08-02T10:29:43.410631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clustering_data['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:29:43.413772Z","iopub.execute_input":"2022-08-02T10:29:43.414689Z","iopub.status.idle":"2022-08-02T10:29:43.426271Z","shell.execute_reply.started":"2022-08-02T10:29:43.414638Z","shell.execute_reply":"2022-08-02T10:29:43.425108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To be continued","metadata":{}}]}