{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:24:40.483214Z","iopub.execute_input":"2022-08-14T04:24:40.483716Z","iopub.status.idle":"2022-08-14T04:24:40.494242Z","shell.execute_reply.started":"2022-08-14T04:24:40.483678Z","shell.execute_reply":"2022-08-14T04:24:40.492801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Library import","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport missingno as msno\nimport lightgbm as lgb\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:42.141581Z","iopub.execute_input":"2022-08-14T04:24:42.142343Z","iopub.status.idle":"2022-08-14T04:24:43.933782Z","shell.execute_reply.started":"2022-08-14T04:24:42.142273Z","shell.execute_reply":"2022-08-14T04:24:43.932677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')\nsub = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:43.935678Z","iopub.execute_input":"2022-08-14T04:24:43.936024Z","iopub.status.idle":"2022-08-14T04:24:44.247260Z","shell.execute_reply.started":"2022-08-14T04:24:43.935994Z","shell.execute_reply":"2022-08-14T04:24:44.245944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:45.490427Z","iopub.execute_input":"2022-08-14T04:24:45.490950Z","iopub.status.idle":"2022-08-14T04:24:45.536421Z","shell.execute_reply.started":"2022-08-14T04:24:45.490907Z","shell.execute_reply":"2022-08-14T04:24:45.535031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:45.701811Z","iopub.execute_input":"2022-08-14T04:24:45.702323Z","iopub.status.idle":"2022-08-14T04:24:45.759606Z","shell.execute_reply.started":"2022-08-14T04:24:45.702284Z","shell.execute_reply":"2022-08-14T04:24:45.758295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:46.577669Z","iopub.execute_input":"2022-08-14T04:24:46.578150Z","iopub.status.idle":"2022-08-14T04:24:46.592648Z","shell.execute_reply.started":"2022-08-14T04:24:46.578110Z","shell.execute_reply":"2022-08-14T04:24:46.591790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:47.600868Z","iopub.execute_input":"2022-08-14T04:24:47.602117Z","iopub.status.idle":"2022-08-14T04:24:48.564586Z","shell.execute_reply.started":"2022-08-14T04:24:47.602072Z","shell.execute_reply":"2022-08-14T04:24:48.563249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**insight**   \nThere are no duplication.    \nThere are some missing value. Missing value appear to be random.    \nデータに重複はなく、いくつか欠損値が存在する。またそれらの欠損値はランダムに発生しているようにみえる","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:49.736303Z","iopub.execute_input":"2022-08-14T04:24:49.736800Z","iopub.status.idle":"2022-08-14T04:24:49.768550Z","shell.execute_reply.started":"2022-08-14T04:24:49.736761Z","shell.execute_reply":"2022-08-14T04:24:49.767134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_temp = train.copy()\ntrain_temp['missing_value_counts']=train_temp.isnull().sum(axis=1)\ntemp = pd.DataFrame(train_temp['missing_value_counts'].value_counts().sort_index())\nplt.bar(temp.index, temp['missing_value_counts'])\nplt.title('histgram of missing value counts')\nplt.xlabel('missing value counts')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:51.624815Z","iopub.execute_input":"2022-08-14T04:24:51.625379Z","iopub.status.idle":"2022-08-14T04:24:51.824795Z","shell.execute_reply.started":"2022-08-14T04:24:51.625330Z","shell.execute_reply":"2022-08-14T04:24:51.823711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(train_temp.groupby('missing_value_counts').mean()['failure'].index, train_temp.groupby('missing_value_counts').mean()['failure'])\nplt.title('how the failure probability depends on missing value counts')\nplt.xlabel('missing value counts')\nplt.ylabel('failure probability')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:52.014510Z","iopub.execute_input":"2022-08-14T04:24:52.015803Z","iopub.status.idle":"2022-08-14T04:24:52.272144Z","shell.execute_reply.started":"2022-08-14T04:24:52.015749Z","shell.execute_reply":"2022-08-14T04:24:52.269896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**insight**   \nA histogram of the number of missing values shows that there are rows with up to 6 missing values.  \nIt also appears that the number of missing values increases the failure probability.      \nIt is possible that the presence of missing values influences the decision to fail.   \n欠損値の数とデータ数をヒストグラムにしてみると、最大６個の欠損値を持つ行があることがわかる。    \nまた欠損値の数が多いと、failure probabilityが上昇するようにもみえる。   \n欠損値があることがfailureにすることに影響を与えている可能性がある。","metadata":{}},{"cell_type":"markdown","source":"# target","metadata":{}},{"cell_type":"code","source":"plt.bar(train['failure'].value_counts().index.astype(str), train['failure'].value_counts())\nplt.title('failure')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:55.567143Z","iopub.execute_input":"2022-08-14T04:24:55.567696Z","iopub.status.idle":"2022-08-14T04:24:55.760556Z","shell.execute_reply.started":"2022-08-14T04:24:55.567656Z","shell.execute_reply":"2022-08-14T04:24:55.759464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**insight**   \nThe data set is a little imbalanced data.   \n今回のデータは若干不均衡データである。","metadata":{}},{"cell_type":"markdown","source":"# float features","metadata":{}},{"cell_type":"code","source":"float_features = [i for i in train.columns if train[i].dtypes == 'float64']","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:57.709227Z","iopub.execute_input":"2022-08-14T04:24:57.710145Z","iopub.status.idle":"2022-08-14T04:24:57.717297Z","shell.execute_reply.started":"2022-08-14T04:24:57.710094Z","shell.execute_reply":"2022-08-14T04:24:57.716296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize =(12,12))\nsns.heatmap(train[float_features + ['failure']].corr(), center=0, annot=True, fmt='.1f')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:24:57.861823Z","iopub.execute_input":"2022-08-14T04:24:57.862295Z","iopub.status.idle":"2022-08-14T04:24:59.456096Z","shell.execute_reply.started":"2022-08-14T04:24:57.862256Z","shell.execute_reply":"2022-08-14T04:24:59.455225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig1 = plt.figure(figsize=(18,16))\nfig1.suptitle(' histgram of the float features', fontsize =16, y=0.95)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, column in enumerate(float_features):\n    plt.subplot(4, 4, i+1)\n    plt.hist(train[column], bins=100, alpha=0.5, label='train')\n    plt.hist(test[column], bins=100, alpha=0.5, label='test')\n    plt.ylabel('count')\n    plt.title(f'{column}_train_std0 : {train[column].std():.1f}\\n{column}_test_std0 : {test[column].std():.1f}')\n    plt.legend()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T04:25:00.722079Z","iopub.execute_input":"2022-08-14T04:25:00.723323Z","iopub.status.idle":"2022-08-14T04:25:09.715340Z","shell.execute_reply.started":"2022-08-14T04:25:00.723279Z","shell.execute_reply":"2022-08-14T04:25:09.714007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_proba_list = []\nfig = plt.figure(figsize=(18,16))\nfig.suptitle('how the failure probability depends on single features', fontsize =16, y=0.95)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\ny = train['failure'].mean()\nfor i, f in enumerate(float_features):\n    plt.subplot(4, 4, i+1)  \n    temp = pd.DataFrame({f:train[f].values,\n                       'failure':train.failure.values})\n    temp = temp.sort_values(f)\n    temp.reset_index(inplace=True)\n    plt.scatter(temp[f], temp.failure.rolling(15000, center=True).mean(), s=2)\n    plt.axhline(y=y, color = 'red', label='whole average')\n    n = len(train['failure'].loc[train[f].isnull()])\n\n    plt.title(f'{f}\\nmissing value counts:{n}')\n    plt.ylabel('failure probability')\n    plt.ylim(0.16, 0.26)\n    y2 = train['failure'].loc[train[f].isnull()].mean()    \n    plt.axhline(y=y2, color = 'blue', label='average in missing values')\n    plt.legend()\n\n    fail = train.loc[train[f].isnull()].failure.sum()\n    z = (fail - n*y)/(np.sqrt(n*y*(1-y)))\n    missing_proba_list.append([f, y2, z])\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T05:01:22.498492Z","iopub.execute_input":"2022-08-14T05:01:22.499058Z","iopub.status.idle":"2022-08-14T05:01:28.004492Z","shell.execute_reply.started":"2022-08-14T05:01:22.499015Z","shell.execute_reply":"2022-08-14T05:01:28.003294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**insight**   \nWe checked the impact of each float feature on the failure probability.       \nThe red line shows the probability of the entire train data,    \nand the blue line shows the probability of the data with missing values in each column.      \n'loading', 'measurement_3', 'measurement_4', 'measurement_5', and 'measurement_9' appear to be affected by missing values to failure.   \nLet's run a significance test on these columns.   \n各float featuresのfailure probabilityへの影響を確認しました。    \n赤線はtrain data全体のprobability、青線は各カラムの欠損値があるデータのprobabilityを示します。   \n'loading', 'measurement_3', 'measurement_4', 'measurement_5', と 'measurement_9'は差があるように見えます。   \nもう少し深く見てみるために、有意差検定をおこなってみます。","metadata":{}},{"cell_type":"code","source":"missing_proba_df = pd.DataFrame(missing_proba_list, columns=['column_name', 'failure_probability', 'z_score'])\ndef color_background_lightgreen(val):\n    color = '' if -1.96 < val < 1.96 else 'lightgreen' #1より大なら薄緑、その他は白\n    return 'background-color: %s' % color\n\nmissing_proba_df.style.applymap(color_background_lightgreen, subset=['z_score'])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:50:43.383925Z","iopub.execute_input":"2022-08-14T04:50:43.385132Z","iopub.status.idle":"2022-08-14T04:50:43.467742Z","shell.execute_reply.started":"2022-08-14T04:50:43.385076Z","shell.execute_reply":"2022-08-14T04:50:43.465992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**insight**\nSince the z_score of measurement_3 and measurement_5 is greater than ±1.96, we can say that there is a significant difference.   \nmeasurement_3とmeasurement_5はZスコアが±1.96を超えているため、有意差があると言えます。","metadata":{}},{"cell_type":"markdown","source":"# Int features","metadata":{}},{"cell_type":"code","source":"int_features = [i for i in train.columns if train[i].dtypes == 'int']\nint_features.remove('id')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:09:31.036890Z","iopub.execute_input":"2022-08-14T05:09:31.037989Z","iopub.status.idle":"2022-08-14T05:09:31.044719Z","shell.execute_reply.started":"2022-08-14T05:09:31.037937Z","shell.execute_reply":"2022-08-14T05:09:31.043512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize =(12,12))\nsns.heatmap(train[int_features].corr(), center=0, annot=True, fmt='.1f')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:09:31.357332Z","iopub.execute_input":"2022-08-14T05:09:31.358263Z","iopub.status.idle":"2022-08-14T05:09:31.783280Z","shell.execute_reply.started":"2022-08-14T05:09:31.358218Z","shell.execute_reply":"2022-08-14T05:09:31.782030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"attribute_2, attribute_3, measurement_0 and measurement_1 are correlated.","metadata":{}},{"cell_type":"code","source":"int_features.remove('failure')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T05:09:37.708274Z","iopub.execute_input":"2022-08-14T05:09:37.708797Z","iopub.status.idle":"2022-08-14T05:09:37.715106Z","shell.execute_reply.started":"2022-08-14T05:09:37.708758Z","shell.execute_reply":"2022-08-14T05:09:37.713768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(20,5))\nfig.suptitle(' histgram of the int features', fontsize =16)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, column in enumerate(int_features):\n    plt.subplot(1, 5, i+1)\n    plt.hist(train[column], bins=20, alpha=0.5, label='train')\n    plt.hist(test[column], bins=20, alpha=0.5, label='test')\n    plt.title(f'{column}')\n    plt.legend()\nplt.show()\n\nfig = plt.figure(figsize=(20,5))\nfig.suptitle('how the failure probability depends on single features', fontsize =16)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, column in enumerate(int_features):\n    plt.subplot(1, 5, i+1)  \n    plt.bar(train.groupby(column).mean()['failure'].index, train.groupby(column).mean()['failure'])\n    plt.axhline(y=y, color = 'red')\n    plt.ylabel('failure probability')\n    plt.title(f'{column}')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T05:09:39.778655Z","iopub.execute_input":"2022-08-14T05:09:39.779494Z","iopub.status.idle":"2022-08-14T05:09:42.184974Z","shell.execute_reply.started":"2022-08-14T05:09:39.779438Z","shell.execute_reply":"2022-08-14T05:09:42.183932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Object features","metadata":{}},{"cell_type":"code","source":"obj_features = [i for i in train.columns if train[i].dtypes == 'object']","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:14:07.611326Z","iopub.execute_input":"2022-08-14T05:14:07.611855Z","iopub.status.idle":"2022-08-14T05:14:07.619063Z","shell.execute_reply.started":"2022-08-14T05:14:07.611812Z","shell.execute_reply":"2022-08-14T05:14:07.617572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,4))\nfig.suptitle(' histgram of the obj features', fontsize =16, y=0.98)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, column in enumerate(obj_features):\n    plt.subplot(1, 3, i+1) \n    index_list = list(set([i for i in train[column]] + [i for i in test[column]]))\n    index_list.sort()\n    temp_df = pd.DataFrame(index=index_list, columns=['train', 'test'])\n    temp_train = pd.DataFrame(train[column].value_counts().sort_index())\n    temp_test = pd.DataFrame(test[column].value_counts().sort_index())\n    temp_df['train'] = temp_train\n    temp_df['test'] = temp_test\n    plt.bar(temp_df.index, temp_df['train'], alpha=0.5, label='train')\n    plt.bar(temp_df.index, temp_df['test'], alpha=0.5, label='test')\n    plt.legend()\n    plt.ylabel('count')\n    plt.title(f'{column}')\nplt.show()\n\nfig = plt.figure(figsize=(12,4))\nfig.suptitle('how the failure probability depends on single features', fontsize =16)\nplt.subplots_adjust(wspace=0.4, hspace=0.3)\nfor i, column in enumerate(obj_features):\n    plt.subplot(1, 3, i+1)  \n    plt.bar(train.groupby(column).mean()['failure'].index, train.groupby(column).mean()['failure'])\n    plt.axhline(y=y, color = 'red')\n    plt.ylabel('failure probability')\n    plt.title(f'{column}')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-14T05:14:07.848632Z","iopub.execute_input":"2022-08-14T05:14:07.849184Z","iopub.status.idle":"2022-08-14T05:14:08.905546Z","shell.execute_reply.started":"2022-08-14T05:14:07.849132Z","shell.execute_reply":"2022-08-14T05:14:08.904268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combination of attribute","metadata":{}},{"cell_type":"code","source":"attribute_train_df = pd.DataFrame(train.groupby(['product_code', 'attribute_0', 'attribute_1', 'attribute_2', 'attribute_3']).size(), columns=['train'])\nattribute_train_df","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:14:08.907163Z","iopub.execute_input":"2022-08-14T05:14:08.907554Z","iopub.status.idle":"2022-08-14T05:14:08.932401Z","shell.execute_reply.started":"2022-08-14T05:14:08.907520Z","shell.execute_reply":"2022-08-14T05:14:08.931254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attribute_test_df = pd.DataFrame(test.groupby(['product_code', 'attribute_0', 'attribute_1', 'attribute_2', 'attribute_3']).size(), columns=['test'])\nattribute_test_df","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:14:08.974827Z","iopub.execute_input":"2022-08-14T05:14:08.975865Z","iopub.status.idle":"2022-08-14T05:14:08.998314Z","shell.execute_reply.started":"2022-08-14T05:14:08.975816Z","shell.execute_reply":"2022-08-14T05:14:08.997133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**insight**    \nThe product_code is determined by a combination of attributes!!","metadata":{}}]}