{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T13:44:41.535060Z","iopub.execute_input":"2022-08-06T13:44:41.535601Z","iopub.status.idle":"2022-08-06T13:44:41.571915Z","shell.execute_reply.started":"2022-08-06T13:44:41.535493Z","shell.execute_reply":"2022-08-06T13:44:41.571026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nimport seaborn as sns\n\nfrom sklearn import metrics,model_selection,preprocessing,tree\nfrom sklearn import tree,ensemble,decomposition\nfrom sklearn import base,cluster,linear_model\nimport xgboost as xgb\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\n\nfrom scipy.stats import boxcox\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nseed = 13","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:41.573861Z","iopub.execute_input":"2022-08-06T13:44:41.574477Z","iopub.status.idle":"2022-08-06T13:44:50.364815Z","shell.execute_reply.started":"2022-08-06T13:44:41.574444Z","shell.execute_reply":"2022-08-06T13:44:50.363439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root = \"/kaggle/input/tabular-playground-series-aug-2022\"\ntrain = pd.read_csv(f\"{root}/train.csv\")\ntest = pd.read_csv(f\"{root}/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:57.086615Z","iopub.execute_input":"2022-08-06T13:44:57.087487Z","iopub.status.idle":"2022-08-06T13:44:57.371321Z","shell.execute_reply.started":"2022-08-06T13:44:57.087451Z","shell.execute_reply":"2022-08-06T13:44:57.370020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Looking at the data","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:58.221318Z","iopub.execute_input":"2022-08-06T13:44:58.221723Z","iopub.status.idle":"2022-08-06T13:44:58.256736Z","shell.execute_reply.started":"2022-08-06T13:44:58.221689Z","shell.execute_reply":"2022-08-06T13:44:58.255900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:59.284700Z","iopub.execute_input":"2022-08-06T13:44:59.285122Z","iopub.status.idle":"2022-08-06T13:44:59.305959Z","shell.execute_reply.started":"2022-08-06T13:44:59.285089Z","shell.execute_reply":"2022-08-06T13:44:59.305063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:00.299534Z","iopub.execute_input":"2022-08-06T13:45:00.300436Z","iopub.status.idle":"2022-08-06T13:45:00.308836Z","shell.execute_reply.started":"2022-08-06T13:45:00.300400Z","shell.execute_reply":"2022-08-06T13:45:00.308047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.set_index(\"id\",inplace=True)\ntest.set_index(\"id\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:01.471519Z","iopub.execute_input":"2022-08-06T13:45:01.472401Z","iopub.status.idle":"2022-08-06T13:45:01.478599Z","shell.execute_reply.started":"2022-08-06T13:45:01.472366Z","shell.execute_reply":"2022-08-06T13:45:01.477614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([train.drop(\"failure\",axis=1),test],axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:02.653960Z","iopub.execute_input":"2022-08-06T13:45:02.654663Z","iopub.status.idle":"2022-08-06T13:45:02.684643Z","shell.execute_reply.started":"2022-08-06T13:45:02.654629Z","shell.execute_reply":"2022-08-06T13:45:02.683299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:03.771072Z","iopub.execute_input":"2022-08-06T13:45:03.771742Z","iopub.status.idle":"2022-08-06T13:45:03.832145Z","shell.execute_reply.started":"2022-08-06T13:45:03.771704Z","shell.execute_reply":"2022-08-06T13:45:03.831005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:04.397316Z","iopub.execute_input":"2022-08-06T13:45:04.397997Z","iopub.status.idle":"2022-08-06T13:45:04.530044Z","shell.execute_reply.started":"2022-08-06T13:45:04.397933Z","shell.execute_reply":"2022-08-06T13:45:04.528929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:05.042105Z","iopub.execute_input":"2022-08-06T13:45:05.042501Z","iopub.status.idle":"2022-08-06T13:45:05.070432Z","shell.execute_reply.started":"2022-08-06T13:45:05.042470Z","shell.execute_reply":"2022-08-06T13:45:05.069171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in df.columns:\n    if df[col].dtype in (\"int64\",\"object\"):\n        print(f\"Name: {col}\\nDtype: {df[col].dtype}\\nNunique: {df[col].nunique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:05.912796Z","iopub.execute_input":"2022-08-06T13:45:05.913637Z","iopub.status.idle":"2022-08-06T13:45:05.935090Z","shell.execute_reply.started":"2022-08-06T13:45:05.913597Z","shell.execute_reply":"2022-08-06T13:45:05.934200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = df.select_dtypes(include=[\"int\",\"object\"]).columns\ncont_cols = df.select_dtypes(exclude=[\"int\",\"object\"]).columns","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:06.641006Z","iopub.execute_input":"2022-08-06T13:45:06.641777Z","iopub.status.idle":"2022-08-06T13:45:06.653628Z","shell.execute_reply.started":"2022-08-06T13:45:06.641743Z","shell.execute_reply":"2022-08-06T13:45:06.652488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols,cont_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:07.383105Z","iopub.execute_input":"2022-08-06T13:45:07.383879Z","iopub.status.idle":"2022-08-06T13:45:07.392419Z","shell.execute_reply.started":"2022-08-06T13:45:07.383829Z","shell.execute_reply":"2022-08-06T13:45:07.391020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Light EDA and Data imputation","metadata":{}},{"cell_type":"code","source":"nrows = len(cont_cols)//2\nfig,axes = plt.subplots(nrows=nrows,ncols=2,figsize=(25,30))\n\ni=0\n\nfig.suptitle(\"Hist & KDE plot for continuous values columns\",fontsize=20)\nfor j in range(nrows):\n    for k in range(2):\n        sns.histplot(x=cont_cols[i],data=df,ax=axes[j][k],kde=True)\n        axes[j][k].plot(df[cont_cols[i]].mean(),[1450],label=\"mean\",marker=\"o\",markersize=10.0,c=\"g\")\n        axes[j][k].plot(df[cont_cols[i]].median(),[1450],label=\"median\",marker=\"x\",markersize=10.0,c=\"y\")\n        axes[j][k].legend()\n        i+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:09.033061Z","iopub.execute_input":"2022-08-06T13:45:09.033902Z","iopub.status.idle":"2022-08-06T13:45:20.071321Z","shell.execute_reply.started":"2022-08-06T13:45:09.033865Z","shell.execute_reply":"2022-08-06T13:45:20.070222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Except *loading*, the ```measurement_*``` features seem to follow normal distribution.<br>\nSo it's safe to impute the missing values of these columns with the mean of the distribution","metadata":{}},{"cell_type":"code","source":"bell_shaped_cols = cont_cols[1:].copy()\nbell_shaped_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:20.073442Z","iopub.execute_input":"2022-08-06T13:45:20.074051Z","iopub.status.idle":"2022-08-06T13:45:20.082489Z","shell.execute_reply.started":"2022-08-06T13:45:20.074014Z","shell.execute_reply":"2022-08-06T13:45:20.081294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_vals = df[bell_shaped_cols].mean().values","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:20.084451Z","iopub.execute_input":"2022-08-06T13:45:20.084796Z","iopub.status.idle":"2022-08-06T13:45:20.111737Z","shell.execute_reply.started":"2022-08-06T13:45:20.084766Z","shell.execute_reply":"2022-08-06T13:45:20.110281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col_name,col_mean in zip(bell_shaped_cols,mean_vals):\n    df[col_name] = df[col_name].fillna(col_mean)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:20.115277Z","iopub.execute_input":"2022-08-06T13:45:20.116255Z","iopub.status.idle":"2022-08-06T13:45:20.131039Z","shell.execute_reply.started":"2022-08-06T13:45:20.116206Z","shell.execute_reply":"2022-08-06T13:45:20.130064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:20.133081Z","iopub.execute_input":"2022-08-06T13:45:20.133873Z","iopub.status.idle":"2022-08-06T13:45:20.154419Z","shell.execute_reply.started":"2022-08-06T13:45:20.133829Z","shell.execute_reply":"2022-08-06T13:45:20.153167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.skew()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:20.156573Z","iopub.execute_input":"2022-08-06T13:45:20.156946Z","iopub.status.idle":"2022-08-06T13:45:20.193039Z","shell.execute_reply.started":"2022-08-06T13:45:20.156913Z","shell.execute_reply":"2022-08-06T13:45:20.191843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(nrows=1,ncols=2,figsize=(15,10))\nax[0].set_title(\"Before applying box-cox transformation\")\nsns.histplot(data=df[\"loading\"],kde=True,ax=ax[0])\nax[1].set_title(\"After applying box-cox tranforation\")\nsns.histplot(data=boxcox(df[\"loading\"],0),kde=True,ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:20.194789Z","iopub.execute_input":"2022-08-06T13:45:20.195189Z","iopub.status.idle":"2022-08-06T13:45:21.552998Z","shell.execute_reply.started":"2022-08-06T13:45:20.195155Z","shell.execute_reply":"2022-08-06T13:45:21.551624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"loading\"] = boxcox(df[\"loading\"],0)\ndf[\"loading\"] = df[\"loading\"].fillna(df[\"loading\"].mean())\ndf.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:21.554563Z","iopub.execute_input":"2022-08-06T13:45:21.554878Z","iopub.status.idle":"2022-08-06T13:45:21.579190Z","shell.execute_reply.started":"2022-08-06T13:45:21.554849Z","shell.execute_reply":"2022-08-06T13:45:21.577853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:21.580939Z","iopub.execute_input":"2022-08-06T13:45:21.581312Z","iopub.status.idle":"2022-08-06T13:45:21.588999Z","shell.execute_reply.started":"2022-08-06T13:45:21.581281Z","shell.execute_reply":"2022-08-06T13:45:21.587645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax = plt.subplots(nrows=len(cat_cols)//2,ncols=2,figsize=(25,30))\n\ni = 0\nfor j in range(len(cat_cols)//2):\n    for k in range(2):\n        sns.countplot(x=cat_cols[i],data=df,ax=ax[j][k])\n        i+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:21.592649Z","iopub.execute_input":"2022-08-06T13:45:21.593240Z","iopub.status.idle":"2022-08-06T13:45:23.484975Z","shell.execute_reply.started":"2022-08-06T13:45:21.593204Z","shell.execute_reply":"2022-08-06T13:45:23.483776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"### Embedding","metadata":{}},{"cell_type":"code","source":"embed_cols = [\"product_code\"] + [f\"attribute_{i}\" for i in range(4)] + \\\n                [f\"measurement_{i}\" for i in range(3)]\nembed_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:23.486559Z","iopub.execute_input":"2022-08-06T13:45:23.487203Z","iopub.status.idle":"2022-08-06T13:45:23.495829Z","shell.execute_reply.started":"2022-08-06T13:45:23.487167Z","shell.execute_reply":"2022-08-06T13:45:23.494276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in embed_cols:\n    enc = preprocessing.LabelEncoder()\n    df[col] = enc.fit_transform(df[col])\ndf[embed_cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:23.498055Z","iopub.execute_input":"2022-08-06T13:45:23.498559Z","iopub.status.idle":"2022-08-06T13:45:23.582000Z","shell.execute_reply.started":"2022-08-06T13:45:23.498487Z","shell.execute_reply":"2022-08-06T13:45:23.581092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embed_model = tf.keras.models.Sequential([\n    layers.InputLayer(input_shape=(8,)),\n    layers.Embedding(input_dim=df[embed_cols].max().max()+1,output_dim=5),\n    layers.Flatten()\n])\n\nembed_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:23.584094Z","iopub.execute_input":"2022-08-06T13:45:23.584616Z","iopub.status.idle":"2022-08-06T13:45:23.703304Z","shell.execute_reply.started":"2022-08-06T13:45:23.584582Z","shell.execute_reply":"2022-08-06T13:45:23.701872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"op = embed_model(df[embed_cols].values)\nfor i in range(40):\n    df[f\"embed_col{i}\"] = op.numpy()[:,i]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:23.767706Z","iopub.execute_input":"2022-08-06T13:45:23.768184Z","iopub.status.idle":"2022-08-06T13:45:23.878300Z","shell.execute_reply.started":"2022-08-06T13:45:23.768150Z","shell.execute_reply":"2022-08-06T13:45:23.876885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(embed_cols,axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:24.902703Z","iopub.execute_input":"2022-08-06T13:45:24.903751Z","iopub.status.idle":"2022-08-06T13:45:24.923495Z","shell.execute_reply.started":"2022-08-06T13:45:24.903698Z","shell.execute_reply":"2022-08-06T13:45:24.922524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Assign each instance to a cluster","metadata":{}},{"cell_type":"code","source":"def plot_silhouette(X,y):\n    \n    n_clusters = np.unique(y)\n    silhouette_vals = metrics.silhouette_samples(X,y)\n\n    y_ax_lower,y_ax_upper = 0,0\n    y_ticks = []\n\n    for i,c in enumerate(n_clusters):\n        c_silhouette_vals = silhouette_vals[y==c]\n        c_silhouette_vals.sort()\n        y_ax_upper += len(c_silhouette_vals)\n        color = cm.jet(float(i)/len(n_clusters))\n\n        plt.barh(range(y_ax_lower,y_ax_upper),\n                c_silhouette_vals,\n                color=color,\n                height=1.0,\n                edgecolor=\"none\")\n        y_ax_lower += len(c_silhouette_vals)\n        y_ticks.append((y_ax_lower+y_ax_upper)/2)\n    \n    silhouette_avg = np.mean(silhouette_vals)\n    plt.axvline(silhouette_avg,linestyle=\"--\",color=\"red\")\n    plt.yticks(y_ticks,n_clusters+1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:29.025852Z","iopub.execute_input":"2022-08-06T13:45:29.026257Z","iopub.status.idle":"2022-08-06T13:45:29.035215Z","shell.execute_reply.started":"2022-08-06T13:45:29.026226Z","shell.execute_reply":"2022-08-06T13:45:29.034260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"km = cluster.KMeans(n_clusters=3)\ncluster_pred = km.fit_predict(df)\nplot_silhouette(df,cluster_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:10:46.537568Z","iopub.execute_input":"2022-08-06T12:10:46.538561Z","iopub.status.idle":"2022-08-06T12:12:43.042884Z","shell.execute_reply.started":"2022-08-06T12:10:46.538520Z","shell.execute_reply":"2022-08-06T12:12:43.041456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"km = cluster.KMeans(n_clusters=4)\ncluster_pred = km.fit_predict(df)\nplot_silhouette(df,cluster_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:12:43.045850Z","iopub.execute_input":"2022-08-06T12:12:43.046224Z","iopub.status.idle":"2022-08-06T12:14:37.771828Z","shell.execute_reply.started":"2022-08-06T12:12:43.046189Z","shell.execute_reply":"2022-08-06T12:14:37.770648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"km = cluster.KMeans(n_clusters=5)\ncluster_pred = km.fit_predict(df)\nplot_silhouette(df,cluster_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:14:37.773447Z","iopub.execute_input":"2022-08-06T12:14:37.777376Z","iopub.status.idle":"2022-08-06T12:16:33.433165Z","shell.execute_reply.started":"2022-08-06T12:14:37.777330Z","shell.execute_reply":"2022-08-06T12:16:33.432079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"km = cluster.KMeans(n_clusters=6)\ncluster_pred = km.fit_predict(df)\nplot_silhouette(df,cluster_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:16:33.435494Z","iopub.execute_input":"2022-08-06T12:16:33.435872Z","iopub.status.idle":"2022-08-06T12:18:31.505464Z","shell.execute_reply.started":"2022-08-06T12:16:33.435838Z","shell.execute_reply":"2022-08-06T12:18:31.504097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"km = cluster.KMeans(n_clusters=3)\ncluster_pred = km.fit_predict(df)\ndf[\"cluster\"] = cluster_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:33.150191Z","iopub.execute_input":"2022-08-06T13:45:33.150649Z","iopub.status.idle":"2022-08-06T13:45:35.200262Z","shell.execute_reply.started":"2022-08-06T13:45:33.150616Z","shell.execute_reply":"2022-08-06T13:45:35.198934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Engineering some more features","metadata":{}},{"cell_type":"code","source":"df[\"measurement_mean\"] = df[[f\"measurement_{i}\" for i in range(3,17)]].mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:37.375914Z","iopub.execute_input":"2022-08-06T13:45:37.376401Z","iopub.status.idle":"2022-08-06T13:45:37.390226Z","shell.execute_reply.started":"2022-08-06T13:45:37.376366Z","shell.execute_reply":"2022-08-06T13:45:37.388914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"attr2*3\"] = pd.concat([train.attribute_2,test.attribute_2],axis=0)*pd.concat([train.attribute_3,test.attribute_3],axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:38.646554Z","iopub.execute_input":"2022-08-06T13:45:38.646941Z","iopub.status.idle":"2022-08-06T13:45:38.655861Z","shell.execute_reply.started":"2022-08-06T13:45:38.646910Z","shell.execute_reply":"2022-08-06T13:45:38.654826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Preparation","metadata":{}},{"cell_type":"code","source":"train_df = df.loc[train.index]\ntest_df = df.loc[test.index]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:42.737817Z","iopub.execute_input":"2022-08-06T13:45:42.739040Z","iopub.status.idle":"2022-08-06T13:45:42.763253Z","shell.execute_reply.started":"2022-08-06T13:45:42.738988Z","shell.execute_reply":"2022-08-06T13:45:42.761731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"failure\"] = train.failure\ntrain_df.shape[1],train.shape[1]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:43.308320Z","iopub.execute_input":"2022-08-06T13:45:43.308712Z","iopub.status.idle":"2022-08-06T13:45:43.319057Z","shell.execute_reply.started":"2022-08-06T13:45:43.308681Z","shell.execute_reply":"2022-08-06T13:45:43.317447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X,y = train_df.drop(\"failure\",axis=1),train_df.failure\nX_train,X_test,y_train,y_test = model_selection.train_test_split(X,y,\n                                                                test_size=0.15,\n                                                                random_state=seed,\n                                                                stratify=y)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:44.010153Z","iopub.execute_input":"2022-08-06T13:45:44.011364Z","iopub.status.idle":"2022-08-06T13:45:44.051539Z","shell.execute_reply.started":"2022-08-06T13:45:44.011312Z","shell.execute_reply":"2022-08-06T13:45:44.050383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stratified KFold on multiple models","metadata":{}},{"cell_type":"code","source":"def strat_kfold_results(models,X,y,group=None,n_splits=5,voting=True):\n    skf = model_selection.StratifiedKFold(n_splits=n_splits,\n                                          shuffle=True,\n                                         random_state=seed)\n    fold=1\n    \n    if voting:\n        vtg_clf = ensemble.VotingClassifier([\n            (f\"model_{model.__class__.__name__}\",model) for model in models\n        ],voting=\"soft\")\n        models.append(vtg_clf)\n    \n    for train_ind,val_ind in skf.split(X,y,group):\n        \n        print(f\"{'='*20}FOLD:{fold}{'='*20}\")\n        x_train,x_val = X[train_ind],X[val_ind]\n        y_train,y_val = y[train_ind],y[val_ind]\n        \n        for model in models:\n            model = model.fit(x_train,y_train)\n            pred_proba = model.predict_proba(x_val)[:,1]\n            roc_score = metrics.roc_auc_score(y_val,pred_proba)\n            print(f\"MODEL: {model.__class__.__name__} ROC SCORE: {roc_score}\")\n        fold+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:46.199634Z","iopub.execute_input":"2022-08-06T13:45:46.200053Z","iopub.status.idle":"2022-08-06T13:45:46.209140Z","shell.execute_reply.started":"2022-08-06T13:45:46.200022Z","shell.execute_reply":"2022-08-06T13:45:46.208081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_1 = [\n    xgb.XGBClassifier(n_estimators=200),\n    tree.DecisionTreeClassifier(),\n    ensemble.RandomForestClassifier(n_estimators=200),\n    ensemble.AdaBoostClassifier(n_estimators=200),\n    ensemble.GradientBoostingClassifier(n_estimators=200)\n]\n\nstrat_kfold_results(models=models_1,\n                   X=X_train.values,\n                   y=y_train.values,\n                   n_splits=3)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:25:41.465199Z","iopub.execute_input":"2022-08-06T12:25:41.465734Z","iopub.status.idle":"2022-08-06T12:31:54.801099Z","shell.execute_reply.started":"2022-08-06T12:25:41.465663Z","shell.execute_reply":"2022-08-06T12:31:54.799558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_2 = [\n    xgb.XGBClassifier(n_estimators=300,learning_rate=0.001,random_state=seed),\n    ensemble.RandomForestClassifier(n_estimators=300,random_state=seed),\n    ensemble.AdaBoostClassifier(n_estimators=300,learning_rate=0.001,random_state=seed),\n    ensemble.GradientBoostingClassifier(n_estimators=300,learning_rate=0.001,random_state=seed)\n]\n\nstrat_kfold_results(models=models_2,\n                   X=X_train.values,\n                   y=y_train.values,\n                   n_splits=3,\n                   voting=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:34:20.280060Z","iopub.execute_input":"2022-08-06T12:34:20.280549Z","iopub.status.idle":"2022-08-06T12:38:55.969589Z","shell.execute_reply.started":"2022-08-06T12:34:20.280491Z","shell.execute_reply":"2022-08-06T12:38:55.968299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_3 = [\n    xgb.XGBClassifier(n_estimators=500,learning_rate=0.001,random_state=seed),\n    ensemble.AdaBoostClassifier(n_estimators=500,learning_rate=0.001,random_state=seed),\n    ensemble.GradientBoostingClassifier(n_estimators=500,learning_rate=0.001,random_state=seed)\n]\n\nstrat_kfold_results(models=models_3,\n                   X=X_train.values,\n                   y=y_train.values,\n                   n_splits=3,\n                   voting=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:39:56.242308Z","iopub.execute_input":"2022-08-06T12:39:56.242774Z","iopub.status.idle":"2022-08-06T12:45:43.713860Z","shell.execute_reply.started":"2022-08-06T12:39:56.242732Z","shell.execute_reply":"2022-08-06T12:45:43.711954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_4 = [\n    xgb.XGBClassifier(n_estimators=700,learning_rate=0.007,random_state=seed),\n    ensemble.AdaBoostClassifier(n_estimators=700,learning_rate=0.007,random_state=seed),\n    ensemble.GradientBoostingClassifier(n_estimators=700,learning_rate=0.007,random_state=seed)\n]\n\nstrat_kfold_results(models=models_4,\n                   X=X_train.values,\n                   y=y_train.values,\n                   n_splits=3,\n                   voting=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:45:43.727672Z","iopub.execute_input":"2022-08-06T12:45:43.728353Z","iopub.status.idle":"2022-08-06T12:53:58.713734Z","shell.execute_reply.started":"2022-08-06T12:45:43.728296Z","shell.execute_reply":"2022-08-06T12:53:58.712273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stacking and Blending","metadata":{}},{"cell_type":"code","source":"class StackAndBlend(base.BaseEstimator):\n\n    def __init__(self,models,blender,cv_strategy):\n        \n        super(StackAndBlend,self).__init__()\n        if not isinstance(models,list):\n            models = (models,)\n        \n        self.cv_strategy = cv_strategy\n        self.models = models\n        self.blender = blender\n    \n    def fit(self,X,y):\n\n        val_inds = []\n\n        if isinstance(X,pd.DataFrame):\n            X = X.values\n        if isinstance(y,pd.DataFrame) or isinstance(y,pd.Series):\n            y = y.values\n\n    \n        pred_matrix = np.empty(shape=(len(X),len(self.models)))\n        for train_ind,val_ind in self.cv_strategy.split(X,y):\n            X_train,X_val = X[train_ind],X[val_ind]\n            y_train,y_val = y[train_ind],y[val_ind]\n            \n            \n\n            for i in range(len(self.models)):\n                self.models[i] = self.models[i].fit(X_train,y_train)\n                pred_matrix[len(val_inds):len(val_inds)+len(val_ind),i] = self.models[i].predict_proba(X_val)[:,1]\n            \n            val_inds.extend(list(val_ind))\n            \n        for i in range(len(self.models)):\n            self.models[i] = self.models[i].fit(X,y)\n        \n        pred_df = pd.DataFrame(data=pred_matrix[:len(val_inds),:],columns=[f\"pred_{i}\" for i in range(len(self.models))])\n        pred_df.index = val_inds\n        pred_df.sort_index(inplace=True)\n        val_inds.sort()\n        pred_df[\"target\"] = y[val_inds]\n        self.pred_df = pred_df\n        \n        self.blender = self.blender.fit(self.pred_df.drop(\"target\",axis=1),\n                                        self.pred_df.target)\n        \n        return self\n    \n    def predict(self,X):\n        \n        if isinstance(X,pd.DataFrame):\n            X = X.values\n        \n        pred_matrix = np.empty(shape=(len(X),len(self.models)))\n\n        for i in range(len(self.models)):\n            pred_matrix[:,i] = self.models[i].predict_proba(X)[:,1]\n        \n        y_final_pred = self.blender.predict_proba(pred_matrix)[:,1]\n        return y_final_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:52.488065Z","iopub.execute_input":"2022-08-06T13:45:52.488911Z","iopub.status.idle":"2022-08-06T13:45:52.503935Z","shell.execute_reply.started":"2022-08-06T13:45:52.488877Z","shell.execute_reply":"2022-08-06T13:45:52.502907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_sb = [\n    xgb.XGBClassifier(n_estimators=300,learning_rate=0.001,random_state=seed),\n    ensemble.AdaBoostClassifier(n_estimators=700,learning_rate=0.007,random_state=seed),\n    ensemble.GradientBoostingClassifier(n_estimators=500,learning_rate=0.001,random_state=seed)\n]\n\nsb = StackAndBlend(models=models_sb,\n                  blender=linear_model.LogisticRegression(),\n                  cv_strategy=model_selection.StratifiedKFold(n_splits=10))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:59:41.540321Z","iopub.execute_input":"2022-08-06T12:59:41.541048Z","iopub.status.idle":"2022-08-06T12:59:41.549925Z","shell.execute_reply.started":"2022-08-06T12:59:41.541001Z","shell.execute_reply":"2022-08-06T12:59:41.548768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sb = sb.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:59:52.960910Z","iopub.execute_input":"2022-08-06T12:59:52.961349Z","iopub.status.idle":"2022-08-06T13:30:26.026151Z","shell.execute_reply.started":"2022-08-06T12:59:52.961314Z","shell.execute_reply":"2022-08-06T13:30:26.024584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics.roc_auc_score(y_test,sb.predict(X_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:35:17.568256Z","iopub.execute_input":"2022-08-06T13:35:17.568678Z","iopub.status.idle":"2022-08-06T13:35:18.314120Z","shell.execute_reply.started":"2022-08-06T13:35:17.568641Z","shell.execute_reply":"2022-08-06T13:35:18.312831Z"},"trusted":true},"execution_count":null,"outputs":[]}]}