{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T14:28:45.798346Z","iopub.execute_input":"2022-08-03T14:28:45.798804Z","iopub.status.idle":"2022-08-03T14:28:45.809270Z","shell.execute_reply.started":"2022-08-03T14:28:45.798766Z","shell.execute_reply":"2022-08-03T14:28:45.808151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **EDA**\n\n* [**ML models:**](#sec6)\n    1. [LogisticRegression](#sec1)\n    2. [DummyClassifier](#sec2)\n    3. [KNeighborsClassifier](#sec7)\n    4. [SGDClassifier](#sec3)\n    5. [DecisionTreeClassifier](#sec4)\n    6. [ExtraTreesClassifier](#sec7)\n    7. [RandomForestClassifier](#sec5)\n    8. [AdaBoostClassifier](#sec8)\n    9. [catboost](#sec10)\n    \n    \n* **NN**\n    1. [MLPClassifier](#sec9)\n    2. [Keras](#sec11)\n    ","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:28:48.269059Z","iopub.execute_input":"2022-08-03T14:28:48.270124Z","iopub.status.idle":"2022-08-03T14:28:48.894494Z","shell.execute_reply.started":"2022-08-03T14:28:48.270084Z","shell.execute_reply":"2022-08-03T14:28:48.893374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:28:49.029233Z","iopub.execute_input":"2022-08-03T14:28:49.029661Z","iopub.status.idle":"2022-08-03T14:28:49.314631Z","shell.execute_reply.started":"2022-08-03T14:28:49.029625Z","shell.execute_reply":"2022-08-03T14:28:49.313328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**","metadata":{}},{"cell_type":"markdown","source":"The function analyzes NaN and zeros and data types","metadata":{}},{"cell_type":"code","source":"def nan_zero_info(data):\n    print(f'нулей в датасете найдено\\n:{data.isnull().sum()}\\n')\n    print(f'NaN в датасете:\\n {data.isna().sum()}\\n')\n    print(f\"Типы  в датасете:\\n {data.dtypes}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:14:32.647324Z","iopub.execute_input":"2022-08-03T13:14:32.647948Z","iopub.status.idle":"2022-08-03T13:14:32.652863Z","shell.execute_reply.started":"2022-08-03T13:14:32.647898Z","shell.execute_reply":"2022-08-03T13:14:32.651838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train","metadata":{}},{"cell_type":"code","source":"train.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:14:32.655420Z","iopub.execute_input":"2022-08-03T13:14:32.656147Z","iopub.status.idle":"2022-08-03T13:14:32.703293Z","shell.execute_reply.started":"2022-08-03T13:14:32.656071Z","shell.execute_reply":"2022-08-03T13:14:32.702348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:14:32.704764Z","iopub.execute_input":"2022-08-03T13:14:32.705423Z","iopub.status.idle":"2022-08-03T13:14:32.732630Z","shell.execute_reply.started":"2022-08-03T13:14:32.705390Z","shell.execute_reply":"2022-08-03T13:14:32.731612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the ID is useless","metadata":{}},{"cell_type":"code","source":"train.pop('id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:14:32.734408Z","iopub.execute_input":"2022-08-03T13:14:32.735107Z","iopub.status.idle":"2022-08-03T13:14:32.749223Z","shell.execute_reply.started":"2022-08-03T13:14:32.735071Z","shell.execute_reply":"2022-08-03T13:14:32.747962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:44:57.187627Z","iopub.execute_input":"2022-08-02T10:44:57.188084Z","iopub.status.idle":"2022-08-02T10:44:57.216268Z","shell.execute_reply.started":"2022-08-02T10:44:57.188050Z","shell.execute_reply":"2022-08-02T10:44:57.214950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name_col in train.columns:\n    print(f'The most frequent value in the data ({name_col}): {train[name_col].value_counts().idxmax()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:44:59.833099Z","iopub.execute_input":"2022-08-02T10:44:59.833903Z","iopub.status.idle":"2022-08-02T10:44:59.879817Z","shell.execute_reply.started":"2022-08-02T10:44:59.833860Z","shell.execute_reply":"2022-08-02T10:44:59.878946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['failure'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:01.037005Z","iopub.execute_input":"2022-08-02T10:45:01.037609Z","iopub.status.idle":"2022-08-02T10:45:01.050128Z","shell.execute_reply.started":"2022-08-02T10:45:01.037553Z","shell.execute_reply":"2022-08-02T10:45:01.048386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A slight imbalance","metadata":{}},{"cell_type":"markdown","source":"**Trash in the data**","metadata":{}},{"cell_type":"code","source":"train.index.is_unique, train.index.is_monotonic","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:03.376333Z","iopub.execute_input":"2022-08-02T10:45:03.376816Z","iopub.status.idle":"2022-08-02T10:45:03.385211Z","shell.execute_reply.started":"2022-08-02T10:45:03.376775Z","shell.execute_reply":"2022-08-02T10:45:03.383703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_zero_info(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:04.591986Z","iopub.execute_input":"2022-08-02T10:45:04.593204Z","iopub.status.idle":"2022-08-02T10:45:04.614586Z","shell.execute_reply.started":"2022-08-02T10:45:04.593162Z","shell.execute_reply":"2022-08-02T10:45:04.613076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum().sort_values().plot(kind='barh', figsize=(20,20))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:05.388217Z","iopub.execute_input":"2022-08-02T10:45:05.388964Z","iopub.status.idle":"2022-08-02T10:45:06.708955Z","shell.execute_reply.started":"2022-08-02T10:45:05.388910Z","shell.execute_reply":"2022-08-02T10:45:06.707949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of NaN is a bit relative to the amount of data","metadata":{}},{"cell_type":"code","source":"for measurement_number in range(3,18):\n    print(f'median and mean of {measurement_number}')\n    print(train[f'measurement_{measurement_number}'].median(), train[f'measurement_{measurement_number}'].mean().round(decimals = 2))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:07:16.930082Z","iopub.execute_input":"2022-08-02T11:07:16.930551Z","iopub.status.idle":"2022-08-02T11:07:16.958034Z","shell.execute_reply.started":"2022-08-02T11:07:16.930511Z","shell.execute_reply":"2022-08-02T11:07:16.956915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['loading'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:38.277382Z","iopub.execute_input":"2022-08-02T10:45:38.278219Z","iopub.status.idle":"2022-08-02T10:45:38.291111Z","shell.execute_reply.started":"2022-08-02T10:45:38.278176Z","shell.execute_reply":"2022-08-02T10:45:38.290011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The mean and median are indistinguishable replace NaN with median","metadata":{}},{"cell_type":"code","source":"for measurement_number in range(3,18):\n    train[f'measurement_{measurement_number}'] =\\\n    train[f'measurement_{measurement_number}'].fillna(train[f'measurement_{measurement_number}'].median())\ntrain['loading'] = train['loading'].fillna(train['loading'].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:28:57.413492Z","iopub.execute_input":"2022-08-03T14:28:57.413892Z","iopub.status.idle":"2022-08-03T14:28:57.450300Z","shell.execute_reply.started":"2022-08-03T14:28:57.413862Z","shell.execute_reply":"2022-08-03T14:28:57.449027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:49:04.375047Z","iopub.execute_input":"2022-08-03T07:49:04.375621Z","iopub.status.idle":"2022-08-03T07:49:04.394959Z","shell.execute_reply.started":"2022-08-03T07:49:04.375579Z","shell.execute_reply":"2022-08-03T07:49:04.393776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Statistics**","metadata":{}},{"cell_type":"code","source":"product_code_by_loading = train.groupby('product_code')[\"loading\"].agg([\"mean\", \"median\", \"min\", \"max\", \"count\"])\n\nproduct_code_by_failure = train.groupby('product_code')[\"failure\"].agg([\"mean\", \"median\", \"min\", \"max\", \"count\"])\n\nstats_by_attribute_0 = train.groupby('attribute_0')[\"loading\"].agg([\"mean\", \"median\", \"min\", \"max\", \"count\"])\n\nstats_by_attribute_1 = train.groupby('attribute_1')[\"loading\"].agg([\"mean\", \"median\", \"min\", \"max\", \"count\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:18.011797Z","iopub.execute_input":"2022-08-02T11:08:18.012225Z","iopub.status.idle":"2022-08-02T11:08:18.044144Z","shell.execute_reply.started":"2022-08-02T11:08:18.012187Z","shell.execute_reply":"2022-08-02T11:08:18.043206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"product_code_by_failure","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:19.007421Z","iopub.execute_input":"2022-08-02T11:08:19.007882Z","iopub.status.idle":"2022-08-02T11:08:19.021470Z","shell.execute_reply.started":"2022-08-02T11:08:19.007843Z","shell.execute_reply":"2022-08-02T11:08:19.020275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":":)","metadata":{}},{"cell_type":"code","source":"product_code_by_loading","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:22.111225Z","iopub.execute_input":"2022-08-02T11:08:22.112035Z","iopub.status.idle":"2022-08-02T11:08:22.125336Z","shell.execute_reply.started":"2022-08-02T11:08:22.111983Z","shell.execute_reply":"2022-08-02T11:08:22.124155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_product_code = list(train[\"product_code\"].value_counts())\nlabel_product_code = list(pd.unique(train[\"product_code\"]))\nfig,ax = plt.subplots()\nax.pie(count_product_code, labels=label_product_code, radius=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:25.187245Z","iopub.execute_input":"2022-08-02T11:08:25.187717Z","iopub.status.idle":"2022-08-02T11:08:25.303830Z","shell.execute_reply.started":"2022-08-02T11:08:25.187678Z","shell.execute_reply":"2022-08-02T11:08:25.302638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1 All classes are approximately equivalent\n\n2 Basically the average value by class is 122\n\n3 There are emissions by class","metadata":{}},{"cell_type":"code","source":"stats_by_attribute_0","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:29.439368Z","iopub.execute_input":"2022-08-02T11:08:29.439838Z","iopub.status.idle":"2022-08-02T11:08:29.452770Z","shell.execute_reply.started":"2022-08-02T11:08:29.439799Z","shell.execute_reply":"2022-08-02T11:08:29.451609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_attribute_0 = list(train[\"attribute_0\"].value_counts())\nlabel_attribute_0 = list(pd.unique(train[\"attribute_0\"]))\nfig,ax = plt.subplots()\nax.pie(count_attribute_0, labels=label_attribute_0, radius=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:31.191275Z","iopub.execute_input":"2022-08-02T11:08:31.191691Z","iopub.status.idle":"2022-08-02T11:08:31.312121Z","shell.execute_reply.started":"2022-08-02T11:08:31.191640Z","shell.execute_reply":"2022-08-02T11:08:31.310105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"material number 7 is much more","metadata":{}},{"cell_type":"code","source":"stats_by_attribute_1","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:36.377554Z","iopub.execute_input":"2022-08-02T11:08:36.377995Z","iopub.status.idle":"2022-08-02T11:08:36.391203Z","shell.execute_reply.started":"2022-08-02T11:08:36.377958Z","shell.execute_reply":"2022-08-02T11:08:36.390081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_attribute_1 = list(train[\"attribute_1\"].value_counts())\nlabel_attribute_1 = list(pd.unique(train[\"attribute_1\"]))\nfig,ax = plt.subplots()\nax.pie(count_attribute_1, labels=label_attribute_1, radius=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:39.034270Z","iopub.execute_input":"2022-08-02T11:08:39.035246Z","iopub.status.idle":"2022-08-02T11:08:39.150073Z","shell.execute_reply.started":"2022-08-02T11:08:39.035209Z","shell.execute_reply":"2022-08-02T11:08:39.148791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data = train, x = 'failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:08:42.682970Z","iopub.execute_input":"2022-08-02T11:08:42.684205Z","iopub.status.idle":"2022-08-02T11:08:42.871790Z","shell.execute_reply.started":"2022-08-02T11:08:42.684100Z","shell.execute_reply":"2022-08-02T11:08:42.870675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax1,ax2 = plt.figure(figsize = (25,10)).subplots(1,2)\nsns.histplot(data = train, x = 'loading', ax= ax1)\nsns.countplot(data = train, x = 'failure', ax= ax2)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:30:47.998196Z","iopub.execute_input":"2022-08-02T12:30:47.999921Z","iopub.status.idle":"2022-08-02T12:30:48.614307Z","shell.execute_reply.started":"2022-08-02T12:30:47.999836Z","shell.execute_reply":"2022-08-02T12:30:48.612753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution is approximately normal","metadata":{}},{"cell_type":"markdown","source":"**measurement_data**","metadata":{}},{"cell_type":"code","source":"ax1,ax2 = plt.figure(figsize = (25,10)).subplots(1,2)\nsns.histplot(data = train, x = 'measurement_4', ax= ax1)\nsns.histplot(data = train, x = 'measurement_3', ax= ax2)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:30:53.880337Z","iopub.execute_input":"2022-08-02T12:30:53.880837Z","iopub.status.idle":"2022-08-02T12:30:54.625895Z","shell.execute_reply.started":"2022-08-02T12:30:53.880793Z","shell.execute_reply":"2022-08-02T12:30:54.624311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution is approximately normal","metadata":{}},{"cell_type":"code","source":"measurement_data = train.drop(columns = ['product_code','attribute_0','loading','attribute_1','attribute_2','attribute_3','failure'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:31:01.452376Z","iopub.execute_input":"2022-08-02T12:31:01.453188Z","iopub.status.idle":"2022-08-02T12:31:01.479039Z","shell.execute_reply.started":"2022-08-02T12:31:01.453116Z","shell.execute_reply":"2022-08-02T12:31:01.477450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"measurement_data","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:31:03.436231Z","iopub.execute_input":"2022-08-02T12:31:03.437941Z","iopub.status.idle":"2022-08-02T12:31:03.474564Z","shell.execute_reply.started":"2022-08-02T12:31:03.437874Z","shell.execute_reply":"2022-08-02T12:31:03.473287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in list(measurement_data.columns):\n    ax1= plt.axes()\n    if col in ['measurement_0','measurement_1','measurement_2']:\n        sns.countplot(data=measurement_data, x=col, ax=ax1)\n        ax1.yaxis.grid()\n        ax1.spines['right'].set_visible(False)\n        ax1.spines['top'].set_visible(False)\n        plt.show()\n        \n    sns.histplot(data=measurement_data, x=col, ax=ax1)\n    ax1.yaxis.grid()\n    ax1.spines['right'].set_visible(False)\n    ax1.spines['top'].set_visible(False)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:31:08.892473Z","iopub.execute_input":"2022-08-02T12:31:08.893938Z","iopub.status.idle":"2022-08-02T12:31:15.642848Z","shell.execute_reply.started":"2022-08-02T12:31:08.893873Z","shell.execute_reply":"2022-08-02T12:31:15.641490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost all data features have a normal distribution","metadata":{}},{"cell_type":"code","source":"cat_cols, num_cols = train.dtypes[train.dtypes == 'object'].keys(), train.dtypes[train.dtypes != 'object'].keys()\nfor col in list(num_cols):\n    sns.lineplot(data=train,x=col,y='failure')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:20:17.163434Z","iopub.execute_input":"2022-08-02T11:20:17.163970Z","iopub.status.idle":"2022-08-02T11:43:06.813624Z","shell.execute_reply.started":"2022-08-02T11:20:17.163927Z","shell.execute_reply":"2022-08-02T11:43:06.812417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"attribute_3 dependency on failure as y = 1/x . The dependency graph is falling smoothly","metadata":{}},{"cell_type":"code","source":"train.describe().round(decimals = 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:31:19.402265Z","iopub.execute_input":"2022-08-02T12:31:19.402694Z","iopub.status.idle":"2022-08-02T12:31:19.523169Z","shell.execute_reply.started":"2022-08-02T12:31:19.402659Z","shell.execute_reply":"2022-08-02T12:31:19.521822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr().style.background_gradient(cmap='coolwarm')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:31:22.580022Z","iopub.execute_input":"2022-08-02T12:31:22.580473Z","iopub.status.idle":"2022-08-02T12:31:22.697925Z","shell.execute_reply.started":"2022-08-02T12:31:22.580436Z","shell.execute_reply":"2022-08-02T12:31:22.696595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"failure does not correlate with the signs","metadata":{}},{"cell_type":"markdown","source":"**test**","metadata":{}},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:43:17.948324Z","iopub.execute_input":"2022-08-02T10:43:17.948825Z","iopub.status.idle":"2022-08-02T10:43:18.000813Z","shell.execute_reply.started":"2022-08-02T10:43:17.948777Z","shell.execute_reply":"2022-08-02T10:43:17.999693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.tail(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:43:34.072444Z","iopub.execute_input":"2022-08-02T10:43:34.073013Z","iopub.status.idle":"2022-08-02T10:43:34.107948Z","shell.execute_reply.started":"2022-08-02T10:43:34.072963Z","shell.execute_reply":"2022-08-02T10:43:34.106631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.pop('id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:14:54.617591Z","iopub.execute_input":"2022-08-03T13:14:54.618007Z","iopub.status.idle":"2022-08-03T13:14:54.629915Z","shell.execute_reply.started":"2022-08-03T13:14:54.617968Z","shell.execute_reply":"2022-08-03T13:14:54.629138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_zero_info(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:18.228268Z","iopub.execute_input":"2022-08-02T10:45:18.228748Z","iopub.status.idle":"2022-08-02T10:45:18.247948Z","shell.execute_reply.started":"2022-08-02T10:45:18.228708Z","shell.execute_reply":"2022-08-02T10:45:18.246872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum().sort_values().plot(kind='barh', figsize=(20,20))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:45:22.676595Z","iopub.execute_input":"2022-08-02T10:45:22.678133Z","iopub.status.idle":"2022-08-02T10:45:23.102898Z","shell.execute_reply.started":"2022-08-02T10:45:22.678076Z","shell.execute_reply":"2022-08-02T10:45:23.102019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for measurement_number in range(3,18):\n    test[f'measurement_{measurement_number}'] =\\\n    test[f'measurement_{measurement_number}'].fillna(test[f'measurement_{measurement_number}'].median())\ntest['loading'] = test['loading'].fillna(test['loading'].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:15:06.936627Z","iopub.execute_input":"2022-08-03T13:15:06.937173Z","iopub.status.idle":"2022-08-03T13:15:06.964971Z","shell.execute_reply.started":"2022-08-03T13:15:06.937124Z","shell.execute_reply":"2022-08-03T13:15:06.963919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:46:45.516149Z","iopub.execute_input":"2022-08-02T10:46:45.517085Z","iopub.status.idle":"2022-08-02T10:46:45.531797Z","shell.execute_reply.started":"2022-08-02T10:46:45.517043Z","shell.execute_reply":"2022-08-02T10:46:45.530698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_product_code = list(test[\"product_code\"].value_counts())\nlabel_product_code = list(pd.unique(test[\"product_code\"]))\nfig,ax = plt.subplots()\nax.pie(count_product_code, labels=label_product_code, radius=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:43:21.782706Z","iopub.execute_input":"2022-08-02T11:43:21.783138Z","iopub.status.idle":"2022-08-02T11:43:21.902822Z","shell.execute_reply.started":"2022-08-02T11:43:21.783101Z","shell.execute_reply":"2022-08-02T11:43:21.900922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **ML** <a class=\"anchor\" id=\"sec6\"></a>","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import balanced_accuracy_score\nfrom sklearn.metrics import roc_auc_score,roc_curve\nfrom sklearn.model_selection import GridSearchCV,train_test_split\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import KFold,StratifiedKFold\n\nimport catboost as cat\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier,StackingClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.dummy import DummyClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import SGDClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:29:25.018753Z","iopub.execute_input":"2022-08-03T14:29:25.019445Z","iopub.status.idle":"2022-08-03T14:29:25.750349Z","shell.execute_reply.started":"2022-08-03T14:29:25.019407Z","shell.execute_reply":"2022-08-03T14:29:25.748955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\n\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:15:19.013724Z","iopub.execute_input":"2022-08-03T13:15:19.014158Z","iopub.status.idle":"2022-08-03T13:15:19.064807Z","shell.execute_reply.started":"2022-08-03T13:15:19.014119Z","shell.execute_reply":"2022-08-03T13:15:19.063307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Copying data to experiments","metadata":{}},{"cell_type":"markdown","source":"**dummy** <a class=\"anchor\" id=\"sec2\"></a>","metadata":{}},{"cell_type":"code","source":"dummy_model = DummyClassifier(strategy='stratified')\n\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.3)\n\ndummy_model.fit(X_train,y_train)\n\npred = dummy_model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:47:25.341723Z","iopub.execute_input":"2022-08-02T11:47:25.342818Z","iopub.status.idle":"2022-08-02T11:47:25.361891Z","shell.execute_reply.started":"2022-08-02T11:47:25.342774Z","shell.execute_reply":"2022-08-02T11:47:25.360912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'accuracy: {balanced_accuracy_score(y_test, pred).round(decimals = 2)}, roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:47:26.545445Z","iopub.execute_input":"2022-08-02T11:47:26.546572Z","iopub.status.idle":"2022-08-02T11:47:26.559141Z","shell.execute_reply.started":"2022-08-02T11:47:26.546527Z","shell.execute_reply":"2022-08-02T11:47:26.557987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = y_test\ny_probas = pred\nfpr, tpr, thresholds = roc_curve(y_true, y_probas, pos_label=0)\n\nplt.plot(fpr,tpr)\nplt.show() \n\nauc = np.trapz(tpr,fpr)\nprint('AUC:', auc)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:47:29.433752Z","iopub.execute_input":"2022-08-02T11:47:29.434203Z","iopub.status.idle":"2022-08-02T11:47:29.624276Z","shell.execute_reply.started":"2022-08-02T11:47:29.434164Z","shell.execute_reply":"2022-08-02T11:47:29.623158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's horrible :) .We will start from this model","metadata":{}},{"cell_type":"markdown","source":"**LogisticRegression** <a class=\"anchor\" id=\"sec1\"></a>","metadata":{}},{"cell_type":"code","source":"log_model = LogisticRegression(max_iter=1000)\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, test_size=0.2, stratify=label,random_state=0)\n\nlog_model.fit(X_train, y_train)\npred = log_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:54:40.053155Z","iopub.execute_input":"2022-08-03T07:54:40.053640Z","iopub.status.idle":"2022-08-03T07:54:40.087806Z","shell.execute_reply.started":"2022-08-03T07:54:40.053606Z","shell.execute_reply":"2022-08-03T07:54:40.086266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:47:34.821422Z","iopub.execute_input":"2022-08-02T11:47:34.822271Z","iopub.status.idle":"2022-08-02T11:47:34.836055Z","shell.execute_reply.started":"2022-08-02T11:47:34.822224Z","shell.execute_reply":"2022-08-02T11:47:34.834372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Standard Scaler*","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(data)\n\nstand_data = scaler.transform(data)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:50:52.165716Z","iopub.execute_input":"2022-08-02T11:50:52.166145Z","iopub.status.idle":"2022-08-02T11:50:52.190395Z","shell.execute_reply.started":"2022-08-02T11:50:52.166109Z","shell.execute_reply":"2022-08-02T11:50:52.189351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_model = LogisticRegression(max_iter=1400)\n\nX_train, X_test, y_train, y_test = train_test_split(stand_data, label, test_size=0.33, stratify=label,random_state=0)\nlog_model.fit(X_train, y_train)\npred = log_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:51:04.525596Z","iopub.execute_input":"2022-08-02T11:51:04.526054Z","iopub.status.idle":"2022-08-02T11:51:04.600481Z","shell.execute_reply.started":"2022-08-02T11:51:04.526020Z","shell.execute_reply":"2022-08-02T11:51:04.598831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T11:51:05.629576Z","iopub.execute_input":"2022-08-02T11:51:05.630062Z","iopub.status.idle":"2022-08-02T11:51:05.641916Z","shell.execute_reply.started":"2022-08-02T11:51:05.630020Z","shell.execute_reply":"2022-08-02T11:51:05.640350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Normalization didn't help","metadata":{}},{"cell_type":"markdown","source":"Delete the product code","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nlogist_model = LogisticRegression(max_iter=1200)\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.3)\n\nlogist_model.fit(X_train, y_train)\npred = logist_model.predict_proba(X_test)[:, 1]\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:29:56.220744Z","iopub.execute_input":"2022-08-03T06:29:56.221268Z","iopub.status.idle":"2022-08-03T06:29:57.090008Z","shell.execute_reply.started":"2022-08-03T06:29:56.221226Z","shell.execute_reply":"2022-08-03T06:29:57.088364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata.pop('attribute_1')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nmodern_logistic_model = LogisticRegression(class_weight='balanced')\nnormalize = StandardScaler()\n\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.25)\n\nnormalize.fit(X_train)\nx_train_normalized = normalize.transform(X_train)\nx_test_normalized = normalize.transform(X_test)\n\nparametrs = {\n    'penalty' : ['l1', 'l2', 'elasticnet'],\n    'C' : [0.0001,0.001, 0.01, 0.1, 0.5, 1, 2.5,5],\n    'max_iter': [1000,1500,2000,2200,2500,3500,5000] \n}\n\nlogistic_grid = GridSearchCV(modern_logistic_model, parametrs, cv=5, scoring='roc_auc', n_jobs=-1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:11:33.232813Z","iopub.execute_input":"2022-08-03T08:11:33.233284Z","iopub.status.idle":"2022-08-03T08:11:39.566397Z","shell.execute_reply.started":"2022-08-03T08:11:33.233252Z","shell.execute_reply":"2022-08-03T08:11:39.565149Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(logistic_grid.best_params_)\n#print(roc_auc_score(y_test, logistic_grid.predict_proba(x_test_normalized)[:, 1]))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:11:39.568542Z","iopub.execute_input":"2022-08-03T08:11:39.569294Z","iopub.status.idle":"2022-08-03T08:11:39.590843Z","shell.execute_reply.started":"2022-08-03T08:11:39.569248Z","shell.execute_reply":"2022-08-03T08:11:39.589531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use it!","metadata":{}},{"cell_type":"markdown","source":"Test data score","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata.pop('attribute_1')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\ntest_data = test.copy(deep = True)\ntest_data.pop('product_code')\ntest_data.pop('attribute_1')\ntest_data = pd.get_dummies(test_data)\n\nlogist_model_final = LogisticRegression(C= 0.01, max_iter = 1000, penalty = 'l2')\n\nlogist_model_final.fit(data, label)\n\nprediction = logist_model_final.predict_proba(test_data)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T10:21:02.455447Z","iopub.execute_input":"2022-08-03T10:21:02.455947Z","iopub.status.idle":"2022-08-03T10:21:03.360260Z","shell.execute_reply.started":"2022-08-03T10:21:02.455911Z","shell.execute_reply":"2022-08-03T10:21:03.359236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SGD Classifier** <a class=\"anchor\" id=\"sec3\"></a>","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nsgd_model = SGDClassifier(max_iter=1400,loss = 'hinge',class_weight='balanced')\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, test_size=0.33, stratify=label,random_state=0)\nsgd_model.fit(X_train, y_train)\n\npred = sgd_model.decision_function(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:00:39.309574Z","iopub.execute_input":"2022-08-02T12:00:39.310015Z","iopub.status.idle":"2022-08-02T12:00:39.847522Z","shell.execute_reply.started":"2022-08-02T12:00:39.309978Z","shell.execute_reply":"2022-08-02T12:00:39.845861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:00:42.873220Z","iopub.execute_input":"2022-08-02T12:00:42.873665Z","iopub.status.idle":"2022-08-02T12:00:42.885782Z","shell.execute_reply.started":"2022-08-02T12:00:42.873611Z","shell.execute_reply":"2022-08-02T12:00:42.884342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the result is no better than the last one","metadata":{}},{"cell_type":"markdown","source":"Delete the product code","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nsgd_model_2 = SGDClassifier(max_iter=1400,loss = 'hinge')\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.3)\n\nsgd_model_2.fit(X_train, y_train)\npred = sgd_model_2.decision_function(X_test)\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:36:01.228887Z","iopub.execute_input":"2022-08-03T06:36:01.229382Z","iopub.status.idle":"2022-08-03T06:36:02.529834Z","shell.execute_reply.started":"2022-08-03T06:36:01.229345Z","shell.execute_reply":"2022-08-03T06:36:02.526844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata.pop('attribute_1')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.25)\n\nnormalize.fit(X_train)\nx_train_normalized = normalize.transform(X_train)\nx_test_normalized = normalize.transform(X_test)\n\nmodern_SGDClassifire_model = SGDClassifier(class_weight='balanced')\n\nparametrs = {\n     'penalty' : ['l2', 'l1', 'elasticnet'],\n     'alpha' : [1e-4,5e-3,1e-3,5e-3,1e-2,5e-2,0.1],\n     'max_iter' :[1000,1500,2000,2500,2700,3000] \n}\n\nSGDClassifire_grid = GridSearchCV(modern_SGDClassifire_model, parametrs, cv=5, scoring='roc_auc', n_jobs=-1)\n\nSGDClassifire_grid.fit(x_train_normalized, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:10:53.388764Z","iopub.execute_input":"2022-08-03T08:10:53.389276Z","iopub.status.idle":"2022-08-03T08:11:18.712644Z","shell.execute_reply.started":"2022-08-03T08:10:53.389242Z","shell.execute_reply":"2022-08-03T08:11:18.711703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('best parameters',SGDClassifire_grid.best_params_)\nprint('best score',SGDClassifire_grid.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:11:50.750667Z","iopub.execute_input":"2022-08-03T08:11:50.751185Z","iopub.status.idle":"2022-08-03T08:11:50.757603Z","shell.execute_reply.started":"2022-08-03T08:11:50.751142Z","shell.execute_reply":"2022-08-03T08:11:50.756602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**decision tree** <a class=\"anchor\" id=\"sec4\"></a>","metadata":{}},{"cell_type":"markdown","source":"let's see the importance of features","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\ndecision_tree_model = DecisionTreeClassifier()\nX_train, X_test, y_train, y_test = train_test_split(data, label, test_size=0.3)\ndecision_tree_model.fit(X_train, y_train)\npred = decision_tree_model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:32:32.229142Z","iopub.execute_input":"2022-08-02T12:32:32.229749Z","iopub.status.idle":"2022-08-02T12:32:33.161050Z","shell.execute_reply.started":"2022-08-02T12:32:32.229689Z","shell.execute_reply":"2022-08-02T12:32:33.159729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:32:33.163078Z","iopub.execute_input":"2022-08-02T12:32:33.163554Z","iopub.status.idle":"2022-08-02T12:32:33.175474Z","shell.execute_reply.started":"2022-08-02T12:32:33.163516Z","shell.execute_reply":"2022-08-02T12:32:33.174017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame({'feature':data.columns, 'importance':decision_tree_model.feature_importances_}).sort_values('importance')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:32:35.215925Z","iopub.execute_input":"2022-08-02T12:32:35.216375Z","iopub.status.idle":"2022-08-02T12:32:35.234929Z","shell.execute_reply.started":"2022-08-02T12:32:35.216337Z","shell.execute_reply":"2022-08-02T12:32:35.233450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Categorical features are of low importance.","metadata":{}},{"cell_type":"code","source":"import shap","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:32:36.224387Z","iopub.execute_input":"2022-08-02T12:32:36.225318Z","iopub.status.idle":"2022-08-02T12:32:39.641385Z","shell.execute_reply.started":"2022-08-02T12:32:36.225262Z","shell.execute_reply":"2022-08-02T12:32:39.639940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X, val_X, train_y, val_y = train_test_split(data, label, random_state=1)\nsmall_val_X = val_X.iloc[:150]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:32:39.860046Z","iopub.execute_input":"2022-08-02T12:32:39.860887Z","iopub.status.idle":"2022-08-02T12:32:39.886057Z","shell.execute_reply.started":"2022-08-02T12:32:39.860844Z","shell.execute_reply":"2022-08-02T12:32:39.884694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"explainer = shap.TreeExplainer(decision_tree_model)\nshap_values = explainer.shap_values(small_val_X)\n\nshap.summary_plot(shap_values, small_val_X)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T12:32:40.568301Z","iopub.execute_input":"2022-08-02T12:32:40.569083Z","iopub.status.idle":"2022-08-02T12:32:41.640505Z","shell.execute_reply.started":"2022-08-02T12:32:40.569042Z","shell.execute_reply":"2022-08-02T12:32:41.639202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Real ML**","metadata":{}},{"cell_type":"markdown","source":"**RandomForestClassifier** <a class=\"anchor\" id=\"sec5\"></a>","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nforest_model = RandomForestClassifier()\nX_train, X_test, y_train, y_test = train_test_split(data, label, test_size=0.3)\n\nforest_model.fit(X_train, y_train)\npred = forest_model.predict_proba(X_test)[:, 1]\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:37:40.616851Z","iopub.execute_input":"2022-08-03T06:37:40.617716Z","iopub.status.idle":"2022-08-03T06:37:50.666792Z","shell.execute_reply.started":"2022-08-03T06:37:40.617668Z","shell.execute_reply":"2022-08-03T06:37:50.665448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The model is poorly trained","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nrandom_forest_model = RandomForestClassifier()\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.3)\n\nrandom_forest_model.fit(X_train, y_train)\npred = random_forest_model.predict_proba(X_test)[:, 1]\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:37:18.985122Z","iopub.execute_input":"2022-08-03T06:37:18.985989Z","iopub.status.idle":"2022-08-03T06:37:30.344460Z","shell.execute_reply.started":"2022-08-03T06:37:18.985925Z","shell.execute_reply":"2022-08-03T06:37:30.342523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**StackingClassifier** <a class=\"anchor\" id=\"sec7\"></a>","metadata":{}},{"cell_type":"markdown","source":"Let's do something more powerful","metadata":{}},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:59:34.751147Z","iopub.execute_input":"2022-08-03T08:59:34.752344Z","iopub.status.idle":"2022-08-03T08:59:34.793692Z","shell.execute_reply.started":"2022-08-03T08:59:34.752289Z","shell.execute_reply":"2022-08-03T08:59:34.792338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"estimators = [('lg_1',LogisticRegression(max_iter = 1000,C = 2,class_weight='balanced')),('lg_2',LogisticRegression(max_iter = 2000,C = 0.01 )),\n             ('sgd_1',SGDClassifier(class_weight='balanced',max_iter = 1000, alpha = 0.2)),('sgd_2',SGDClassifier(max_iter = 2000, alpha = 0.02)),\n             ('ds_1',DecisionTreeClassifier(max_depth=4,max_leaf_nodes=6)),('ds_2',DecisionTreeClassifier(max_depth=6,max_leaf_nodes=2,random_state=10)),\n             ('knn_1',KNeighborsClassifier(n_neighbors=5)), ('knn_2',KNeighborsClassifier(n_neighbors=10)),\n             ('rf', RandomForestClassifier(n_estimators=100, random_state=42))]\n\nfinal_model = LogisticRegression(max_iter=1000,class_weight='balanced')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:33:44.503234Z","iopub.execute_input":"2022-08-03T08:33:44.503703Z","iopub.status.idle":"2022-08-03T08:33:44.512380Z","shell.execute_reply.started":"2022-08-03T08:33:44.503669Z","shell.execute_reply":"2022-08-03T08:33:44.511519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1","metadata":{}},{"cell_type":"code","source":"clf_1 = StackingClassifier(estimators=estimators, final_estimator=final_model)\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, stratify=label, random_state=42)\n\nclf_1.fit(X_train, y_train)\n\npred = clf_1.decision_function(X_test)\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:33:47.583319Z","iopub.execute_input":"2022-08-03T08:33:47.583786Z","iopub.status.idle":"2022-08-03T08:35:13.957595Z","shell.execute_reply.started":"2022-08-03T08:33:47.583749Z","shell.execute_reply":"2022-08-03T08:35:13.955860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2","metadata":{}},{"cell_type":"code","source":"estimators = [('rf', RandomForestClassifier(n_estimators=100, random_state=42)),('lg',LogisticRegression(max_iter=2000)),('knn',KNeighborsClassifier(n_neighbors=10)),\n('ds',DecisionTreeClassifier(random_state=42))]\n\nfinal_model = LogisticRegression(max_iter=1000)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:28:56.057525Z","iopub.execute_input":"2022-08-02T10:28:56.058743Z","iopub.status.idle":"2022-08-02T10:28:56.064765Z","shell.execute_reply.started":"2022-08-02T10:28:56.058666Z","shell.execute_reply":"2022-08-02T10:28:56.063592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_2 = StackingClassifier(estimators=estimators, final_estimator=final_model)\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, stratify=label, random_state=42)\n\nclf_2.fit(X_train, y_train)\n\npred = clf_2.predict_proba(X_test)[:, 1]\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:28:58.364728Z","iopub.execute_input":"2022-08-02T10:28:58.365285Z","iopub.status.idle":"2022-08-02T10:30:02.246227Z","shell.execute_reply.started":"2022-08-02T10:28:58.365236Z","shell.execute_reply":"2022-08-02T10:30:02.245325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3","metadata":{}},{"cell_type":"code","source":"estimators = [('lg_1',LogisticRegression(max_iter=2000,random_state=45)),('lg_2',LogisticRegression(max_iter=1000)),\n              ('knn_1',KNeighborsClassifier(n_neighbors=5)), ('knn_2',KNeighborsClassifier(n_neighbors=10)),\n              ('ds_1',DecisionTreeClassifier(max_depth=4,max_leaf_nodes=6)), ('ds_2',DecisionTreeClassifier(max_depth=6,max_leaf_nodes=2,random_state=10))]\n\nfinal_model = cat.CatBoostClassifier(iterations=2000,\n                              depth=4,\n                              learning_rate=0.03,\n                              custom_loss='AUC')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:41:04.580831Z","iopub.execute_input":"2022-08-02T10:41:04.581340Z","iopub.status.idle":"2022-08-02T10:41:04.589483Z","shell.execute_reply.started":"2022-08-02T10:41:04.581294Z","shell.execute_reply":"2022-08-02T10:41:04.588541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_3 = StackingClassifier(estimators=estimators, final_estimator=final_model)\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, stratify=label, random_state=42)\n\nclf_3.fit(X_train, y_train).score(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:41:07.488757Z","iopub.execute_input":"2022-08-02T10:41:07.489179Z","iopub.status.idle":"2022-08-02T10:41:38.997889Z","shell.execute_reply.started":"2022-08-02T10:41:07.489141Z","shell.execute_reply":"2022-08-02T10:41:38.996724Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = clf_3.predict_proba(X_test)[:, 1]\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T10:41:38.999927Z","iopub.execute_input":"2022-08-02T10:41:39.000842Z","iopub.status.idle":"2022-08-02T10:41:44.908327Z","shell.execute_reply.started":"2022-08-02T10:41:39.000807Z","shell.execute_reply":"2022-08-02T10:41:44.906940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"4","metadata":{}},{"cell_type":"code","source":"estimators = [('lg_1',LogisticRegression(max_iter = 1000,C = 2,class_weight='balanced')),('lg_2',LogisticRegression(max_iter = 2000,C = 0.01 )),\n              ('rf_1', RandomForestClassifier(n_estimators=500, random_state=42)), (('rf_2', RandomForestClassifier(n_estimators=1000, random_state=12))),\n             ('ds_1',DecisionTreeClassifier(max_depth=4,max_leaf_nodes=6)),('ds_2',DecisionTreeClassifier(max_depth=6,max_leaf_nodes=2,random_state=10)),\n             ('knn_1',KNeighborsClassifier(n_neighbors=5)), ('knn_2',KNeighborsClassifier(n_neighbors=10))]\n\nfinal_model = LogisticRegression(max_iter=1000)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T10:27:29.787834Z","iopub.execute_input":"2022-08-03T10:27:29.788341Z","iopub.status.idle":"2022-08-03T10:27:29.797865Z","shell.execute_reply.started":"2022-08-03T10:27:29.788305Z","shell.execute_reply":"2022-08-03T10:27:29.796428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_4 = StackingClassifier(estimators=estimators, final_estimator=final_model)\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, stratify=label, random_state=42)\n\nclf_4.fit(X_train, y_train)\n\npred = clf_4.predict_proba(X_test)[:, 1]\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:59:50.282270Z","iopub.execute_input":"2022-08-03T08:59:50.282807Z","iopub.status.idle":"2022-08-03T09:12:31.813712Z","shell.execute_reply.started":"2022-08-03T08:59:50.282767Z","shell.execute_reply":"2022-08-03T09:12:31.802810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"5","metadata":{}},{"cell_type":"code","source":"estimators = [('lg_1',LogisticRegression(max_iter = 1000,C = 2,class_weight='balanced')),('lg_2',LogisticRegression(max_iter = 2000,C = 0.01 )),\n              ('rf_1', RandomForestClassifier(n_estimators=500, random_state=42)), (('rf_2', RandomForestClassifier(n_estimators=1000, random_state=12))),\n             ('ds_1',DecisionTreeClassifier(max_depth=4,max_leaf_nodes=6)),('ds_2',DecisionTreeClassifier(max_depth=6,max_leaf_nodes=2,random_state=10)),\n             ('knn_1',KNeighborsClassifier(n_neighbors=5)), ('knn_2',KNeighborsClassifier(n_neighbors=10)),\n             ('knn_3',KNeighborsClassifier(n_neighbors=15)), ('ext', ExtraTreesClassifier())]\n\nfinal_model = LogisticRegression(max_iter=1000)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T10:33:44.506745Z","iopub.execute_input":"2022-08-03T10:33:44.507165Z","iopub.status.idle":"2022-08-03T10:33:44.515579Z","shell.execute_reply.started":"2022-08-03T10:33:44.507133Z","shell.execute_reply":"2022-08-03T10:33:44.514563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_5 = StackingClassifier(estimators=estimators, final_estimator=final_model)\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, stratify=label, random_state=42)\n\nclf_5.fit(X_train, y_train)\n\npred = clf_2.predict_proba(X_test)[:, 1]\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T10:33:31.679380Z","iopub.execute_input":"2022-08-03T10:33:31.680290Z","iopub.status.idle":"2022-08-03T10:33:37.654172Z","shell.execute_reply.started":"2022-08-03T10:33:31.680237Z","shell.execute_reply":"2022-08-03T10:33:37.652426Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**AdaBoost Classifier** <a class=\"anchor\" id=\"sec8\"></a>","metadata":{}},{"cell_type":"code","source":"ada_model = AdaBoostClassifier(n_estimators=300, random_state=42)\nada_model.fit(X_train, y_train)\npred = ada_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:39:35.428583Z","iopub.execute_input":"2022-08-03T06:39:35.429035Z","iopub.status.idle":"2022-08-03T06:39:50.516785Z","shell.execute_reply.started":"2022-08-03T06:39:35.429001Z","shell.execute_reply":"2022-08-03T06:39:50.515346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:39:50.519658Z","iopub.execute_input":"2022-08-03T06:39:50.520219Z","iopub.status.idle":"2022-08-03T06:39:50.533827Z","shell.execute_reply.started":"2022-08-03T06:39:50.520169Z","shell.execute_reply":"2022-08-03T06:39:50.532129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nada_model = AdaBoostClassifier()\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.3)\n\nrandom_forest_model.fit(X_train, y_train)\npred = random_forest_model.predict_proba(X_test)[:, 1]\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:40:37.332373Z","iopub.execute_input":"2022-08-03T06:40:37.332920Z","iopub.status.idle":"2022-08-03T06:40:48.964902Z","shell.execute_reply.started":"2022-08-03T06:40:37.332883Z","shell.execute_reply":"2022-08-03T06:40:48.963400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's do something more powerful","metadata":{}},{"cell_type":"markdown","source":"**MLP**  <a class=\"anchor\" id=\"sec9\"></a>","metadata":{}},{"cell_type":"code","source":"mlp_model = MLPClassifier(random_state=42, max_iter=4500)\n\ndata = train.copy(deep = True)\ndata.pop('product_code')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\n\nX_train, X_test, y_train, y_test = train_test_split(data,label, test_size = 0.3)\nmlp_model.fit(X_train, y_train)\npred = mlp_model.predict_proba(X_test)[:, 1]\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:19:58.965954Z","iopub.execute_input":"2022-08-03T13:19:58.966403Z","iopub.status.idle":"2022-08-03T13:20:06.376422Z","shell.execute_reply.started":"2022-08-03T13:19:58.966367Z","shell.execute_reply":"2022-08-03T13:20:06.374975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nmlp_model = MLPClassifier(random_state=42, max_iter=1500)\nmlp_model.fit(X_train, y_train)\npred = mlp_model.predict_proba(X_test)[:, 1]\n\nprint(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:20:32.479979Z","iopub.execute_input":"2022-08-03T13:20:32.480379Z","iopub.status.idle":"2022-08-03T13:20:40.394950Z","shell.execute_reply.started":"2022-08-03T13:20:32.480334Z","shell.execute_reply":"2022-08-03T13:20:40.390645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata.pop('attribute_1')\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\ntest_data = test.copy(deep = True)\ntest_data.pop('product_code')\ntest_data.pop('attribute_1')\ntest_data = pd.get_dummies(test_data)\n\nmlp_model_final = MLPClassifier(random_state=42, max_iter=3500)\nmlp_model_final.fit(data, label)\n\nprediction = mlp_model_final.predict_proba(test_data)[:, 1]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T13:21:03.428120Z","iopub.execute_input":"2022-08-03T13:21:03.429121Z","iopub.status.idle":"2022-08-03T13:21:09.297595Z","shell.execute_reply.started":"2022-08-03T13:21:03.429085Z","shell.execute_reply":"2022-08-03T13:21:09.296197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The model proved to be well remembered","metadata":{}},{"cell_type":"markdown","source":"**CatBoost**  <a class=\"anchor\" id=\"sec10\"></a>","metadata":{}},{"cell_type":"code","source":"data_2 = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\nlabel_2 = data_2.pop('failure')\nX_train, X_test, y_train, y_test = train_test_split(data_2, label_2,stratify=label, test_size=0.3)\n\ncat_cols, num_cols = data_2.dtypes[data_2.dtypes == 'object'].keys(), data_2.dtypes[data_2.dtypes != 'object'].keys()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:51:37.473216Z","iopub.execute_input":"2022-08-03T06:51:37.473710Z","iopub.status.idle":"2022-08-03T06:51:37.621765Z","shell.execute_reply.started":"2022-08-03T06:51:37.473674Z","shell.execute_reply":"2022-08-03T06:51:37.620594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"cat_model = cat.CatBoostClassifier(iterations=4000,\n                              depth=3,\n                              learning_rate=0.03,\n                              l2_leaf_reg=4,\n                              custom_loss='AUC',\n                              thread_count=4,\n                              bagging_temperature=1,\n                              use_best_model=True)\n\n\ncat_model.fit(X_train, y_train, \n          eval_set=(X_test, y_test),\n          cat_features=list(cat_cols),\n          plot=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:51:38.775702Z","iopub.execute_input":"2022-08-03T06:51:38.776203Z","iopub.status.idle":"2022-08-03T06:52:36.883882Z","shell.execute_reply.started":"2022-08-03T06:51:38.776165Z","shell.execute_reply":"2022-08-03T06:52:36.882114Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_score(y_test, cat_model.predict_proba(X_test)[:, 1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:52:36.890366Z","iopub.execute_input":"2022-08-03T06:52:36.891097Z","iopub.status.idle":"2022-08-03T06:52:36.916829Z","shell.execute_reply.started":"2022-08-03T06:52:36.891061Z","shell.execute_reply":"2022-08-03T06:52:36.914907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_2.pop('product_code')\ncat_cols = list(cat_cols) \ncat_cols.remove('product_code')\n\ncat_model = cat.CatBoostClassifier(iterations=4000,\n                              depth=3,\n                              learning_rate=0.03,\n                              l2_leaf_reg=4,\n                              custom_loss='AUC',\n                              thread_count=4,\n                              bagging_temperature=1)\n\nX_train, X_test, y_train, y_test = train_test_split(data_2,label_2, test_size = 0.3)\n\ncat_model.fit(X_train, y_train, cat_features=cat_cols)\npred = cat_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:48:40.348321Z","iopub.execute_input":"2022-08-03T06:48:40.348816Z","iopub.status.idle":"2022-08-03T06:49:25.055076Z","shell.execute_reply.started":"2022-08-03T06:48:40.348781Z","shell.execute_reply":"2022-08-03T06:49:25.053894Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'roc_auc : {roc_auc_score(y_test, pred).round(decimals = 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:51:28.279734Z","iopub.execute_input":"2022-08-03T06:51:28.280158Z","iopub.status.idle":"2022-08-03T06:51:28.294879Z","shell.execute_reply.started":"2022-08-03T06:51:28.280127Z","shell.execute_reply":"2022-08-03T06:51:28.293537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\ndata.pop('product_code')\ndata.pop('attribute_1')\nlabel = data.pop('failure')\n\ntest_data = test.copy(deep = True)\ntest_data.pop('product_code')\ntest_data.pop('attribute_1')\n\ncat_model = cat.CatBoostClassifier(iterations=2000,\n                              depth=5,\n                              learning_rate=0.03,\n                              l2_leaf_reg=4,\n                              custom_loss='AUC',\n                              thread_count=4)\n\n\ncat_model.fit(data, label, cat_features=['attribute_0'])\nprediction = cat_model.predict_proba(test_data)[:, 1]\n\nsubmission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\nsubmission.failure = prediction\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:04:06.133394Z","iopub.execute_input":"2022-08-03T11:04:06.134187Z","iopub.status.idle":"2022-08-03T11:04:22.789874Z","shell.execute_reply.started":"2022-08-03T11:04:06.134135Z","shell.execute_reply":"2022-08-03T11:04:22.788688Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Keras** <a class=\"anchor\" id=\"sec11\"></a>","metadata":{}},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:38:43.485287Z","iopub.execute_input":"2022-08-03T15:38:43.485725Z","iopub.status.idle":"2022-08-03T15:38:43.491598Z","shell.execute_reply.started":"2022-08-03T15:38:43.485688Z","shell.execute_reply":"2022-08-03T15:38:43.490208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.copy(deep = True)\n\ndata = pd.get_dummies(data)\nlabel = data.pop('failure')\n\nX_train, X_test, y_train, y_test = train_test_split(data, label,stratify=label, test_size=0.3)\nX_train = X_train.values\nX_test = X_test.values\ny_train = y_train.values\ny_test = y_test.values","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:38:44.516559Z","iopub.execute_input":"2022-08-03T15:38:44.517288Z","iopub.status.idle":"2022-08-03T15:38:44.570557Z","shell.execute_reply.started":"2022-08-03T15:38:44.517245Z","shell.execute_reply":"2022-08-03T15:38:44.569299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Dense(180, input_dim=31, activation='relu'))\nmodel.add(Dense(60, activation='relu'))\nmodel.add(Dense(2, activation='softmax'))\n\nmodel.compile(loss='sparse_categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nmodel.fit(X_train, y_train, epochs=500, batch_size=50)\n\npred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:38:48.336781Z","iopub.execute_input":"2022-08-03T15:38:48.337195Z","iopub.status.idle":"2022-08-03T15:38:53.394918Z","shell.execute_reply.started":"2022-08-03T15:38:48.337163Z","shell.execute_reply":"2022-08-03T15:38:53.393088Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.evaluate(X_test, y_test)\nprint(\"\\nAccuracy: %.2f%%\" % (scores[1]*100))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:30:44.270144Z","iopub.execute_input":"2022-08-03T14:30:44.270620Z","iopub.status.idle":"2022-08-03T14:30:44.877346Z","shell.execute_reply.started":"2022-08-03T14:30:44.270575Z","shell.execute_reply":"2022-08-03T14:30:44.875973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tests","metadata":{}},{"cell_type":"code","source":"\"\"\"\n\nmodels = []\n\nX_train, X_test, y_train, y_test = train_test_split(data, label, test_size=0.3, random_state=42)\n\nresults = [] \nnames = [] \nscoring = '' \n\nfor name, model in models: \n    kfold = KFold(n_splits=10)\n    cv_results = cross_val_score(model, X_train, y_train, cv=kfold, scoring=scoring)\n    results.append(cv_results) \n    names.append(name)\n\nplt.figure(figsize=(15, 10)) \nsns.set_style(style='whitegrid') \ninitial = sns.boxplot(data=results, palette='Set1', notch=False) \ninitial.set_xticklabels(['']) \ninitial.legend()\nplt.show()\n\n\"\"\"","metadata":{},"execution_count":null,"outputs":[]}]}