{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T15:58:51.557879Z","iopub.execute_input":"2022-08-13T15:58:51.558688Z","iopub.status.idle":"2022-08-13T15:58:52.717715Z","shell.execute_reply.started":"2022-08-13T15:58:51.558578Z","shell.execute_reply":"2022-08-13T15:58:52.716472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('seaborn')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:52.719347Z","iopub.execute_input":"2022-08-13T15:58:52.719693Z","iopub.status.idle":"2022-08-13T15:58:52.725989Z","shell.execute_reply.started":"2022-08-13T15:58:52.719662Z","shell.execute_reply":"2022-08-13T15:58:52.724450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TODO\n* Please note, this is still very much a WIP....\n- [ ] Feature scaling/standardization\n- [ ] More advanced imputation techniques?\n- [ ] Class imbalance problem?\n- [ ] Add dummy features for when NaNs are filled\n\n# References\n* https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense/notebook","metadata":{}},{"cell_type":"markdown","source":"# Reading in the data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ndf_test = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:52.727643Z","iopub.execute_input":"2022-08-13T15:58:52.728593Z","iopub.status.idle":"2022-08-13T15:58:53.042845Z","shell.execute_reply.started":"2022-08-13T15:58:52.728557Z","shell.execute_reply":"2022-08-13T15:58:53.041539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.047753Z","iopub.execute_input":"2022-08-13T15:58:53.048516Z","iopub.status.idle":"2022-08-13T15:58:53.055551Z","shell.execute_reply.started":"2022-08-13T15:58:53.048457Z","shell.execute_reply":"2022-08-13T15:58:53.054257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.057809Z","iopub.execute_input":"2022-08-13T15:58:53.058649Z","iopub.status.idle":"2022-08-13T15:58:53.114651Z","shell.execute_reply.started":"2022-08-13T15:58:53.058603Z","shell.execute_reply":"2022-08-13T15:58:53.113434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.116503Z","iopub.execute_input":"2022-08-13T15:58:53.117319Z","iopub.status.idle":"2022-08-13T15:58:53.164287Z","shell.execute_reply.started":"2022-08-13T15:58:53.117273Z","shell.execute_reply":"2022-08-13T15:58:53.163163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.165876Z","iopub.execute_input":"2022-08-13T15:58:53.166751Z","iopub.status.idle":"2022-08-13T15:58:53.181167Z","shell.execute_reply.started":"2022-08-13T15:58:53.166701Z","shell.execute_reply":"2022-08-13T15:58:53.180123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.182676Z","iopub.execute_input":"2022-08-13T15:58:53.183316Z","iopub.status.idle":"2022-08-13T15:58:53.197127Z","shell.execute_reply.started":"2022-08-13T15:58:53.183283Z","shell.execute_reply":"2022-08-13T15:58:53.195698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['id'], axis='columns')\ndf_test = df_test.drop(['id'], axis='columns')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.198814Z","iopub.execute_input":"2022-08-13T15:58:53.199476Z","iopub.status.idle":"2022-08-13T15:58:53.213252Z","shell.execute_reply.started":"2022-08-13T15:58:53.199431Z","shell.execute_reply":"2022-08-13T15:58:53.212005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot numerical columns","metadata":{}},{"cell_type":"code","source":"cols_numeric = [ f'measurement_{i}' for i in range(18)]\ncols_numeric.extend([\n    'loading'\n    #'attribute_0',\n    #'attribute_1',\n    #'attribute_2',\n    #'attribute_3',\n])\nprint(len(cols_numeric))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.217402Z","iopub.execute_input":"2022-08-13T15:58:53.217799Z","iopub.status.idle":"2022-08-13T15:58:53.225795Z","shell.execute_reply.started":"2022-08-13T15:58:53.217759Z","shell.execute_reply":"2022-08-13T15:58:53.224406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split by failure","metadata":{}},{"cell_type":"code","source":"nrows, ncols = 5, 4\nfig, ax = plt.subplots(nrows, ncols, figsize=(20, 20))\nax = ax.ravel()\nfor i, var in enumerate(cols_numeric):\n    sns.histplot(data=df_train, x=var, hue='failure', ax=ax[i])\n    ax[i].set_label(var)\nax = ax.reshape(nrows, ncols)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:58:53.227560Z","iopub.execute_input":"2022-08-13T15:58:53.228779Z","iopub.status.idle":"2022-08-13T15:59:04.453076Z","shell.execute_reply.started":"2022-08-13T15:58:53.228702Z","shell.execute_reply":"2022-08-13T15:59:04.451923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ratio plots","metadata":{}},{"cell_type":"code","source":"meas_cols = [f'measurement_{i}' for i in range(18)]\nmeas_cols.extend(['failure'])\ndf_train_copy = df_train[meas_cols].copy(deep=True)\nfor var in df_train_copy.columns:\n    if var == 'failure':\n        continue\n    df_train_copy[var] /= df_train['loading']\ndf_train_copy","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:03:04.002578Z","iopub.execute_input":"2022-08-13T16:03:04.003004Z","iopub.status.idle":"2022-08-13T16:03:04.049328Z","shell.execute_reply.started":"2022-08-13T16:03:04.002968Z","shell.execute_reply":"2022-08-13T16:03:04.048169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nrows, ncols = 5, 4\nfig, ax = plt.subplots(nrows, ncols, figsize=(20, 20))\nax = ax.ravel()\nfor i, var in enumerate(meas_cols):\n    sns.histplot(data=df_train_copy, x=var, hue='failure', ax=ax[i])\n    ax[i].set_label(var)\nax = ax.reshape(nrows, ncols)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:03:21.169012Z","iopub.execute_input":"2022-08-13T16:03:21.169387Z","iopub.status.idle":"2022-08-13T16:03:33.484637Z","shell.execute_reply.started":"2022-08-13T16:03:21.169355Z","shell.execute_reply":"2022-08-13T16:03:33.483269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## split by product code","metadata":{}},{"cell_type":"code","source":"nrows, ncols = 5, 4\nfig, ax = plt.subplots(nrows, ncols, figsize=(20, 20))\nax = ax.ravel()\nfor i, var in enumerate(cols_numeric):\n    sns.histplot(data=df_train, x=var, hue='product_code', ax=ax[i], fill=False)\n    ax[i].set_label(var)\nax = ax.reshape(nrows, ncols)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:04.454251Z","iopub.execute_input":"2022-08-13T15:59:04.454559Z","iopub.status.idle":"2022-08-13T15:59:26.881442Z","shell.execute_reply.started":"2022-08-13T15:59:04.454530Z","shell.execute_reply":"2022-08-13T15:59:26.880158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA?\n* Measurements 3-17 seem similar, Gaussians with different mean/std\n* Is it plausible to reduce feature space?","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nmeasurements = [f'measurement_{v}' for v in range(3, 18)]\ndf_copy = df_train[measurements].copy(deep=True)\n\ndf_copy[measurements] = SimpleImputer().fit_transform(df_copy[measurements])\ndf_copy[measurements] = StandardScaler().fit_transform(df_copy[measurements])\n\nN_COMPONENTS = df_copy.shape[1]\npca = PCA(n_components=N_COMPONENTS)\ndata_red = pca.fit_transform(df_copy[measurements])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:26.883089Z","iopub.execute_input":"2022-08-13T15:59:26.883423Z","iopub.status.idle":"2022-08-13T15:59:27.208240Z","shell.execute_reply.started":"2022-08-13T15:59:26.883393Z","shell.execute_reply":"2022-08-13T15:59:27.206522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot cumulative explained variance ratio against n_components\nfig, ax = plt.subplots(figsize=(15, 7))\ncumulative_sum = np.cumsum(pca.explained_variance_ratio_)\nax.plot(range(1, 1 + len(cumulative_sum)), cumulative_sum, marker='o', linestyle='-', color='b')\nax.axhline(y=1, color='r', linestyle='dashed')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:27.210283Z","iopub.execute_input":"2022-08-13T15:59:27.211073Z","iopub.status.idle":"2022-08-13T15:59:27.477532Z","shell.execute_reply.started":"2022-08-13T15:59:27.211024Z","shell.execute_reply":"2022-08-13T15:59:27.476430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# boxplots","metadata":{}},{"cell_type":"code","source":"nrows, ncols = 5, 4\nfig, ax = plt.subplots(nrows, ncols, figsize=(20, 20))\nax = ax.ravel()\nfor i, var in enumerate(cols_numeric):\n    sns.boxplot(x=df_train[var], ax=ax[i])\nax = ax.reshape(nrows, ncols)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:27.479256Z","iopub.execute_input":"2022-08-13T15:59:27.479572Z","iopub.status.idle":"2022-08-13T15:59:29.808865Z","shell.execute_reply.started":"2022-08-13T15:59:27.479542Z","shell.execute_reply":"2022-08-13T15:59:29.807763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Categorical variables","metadata":{}},{"cell_type":"code","source":"cat_cols = [\n    'attribute_0',\n    'attribute_1',\n    'attribute_2',\n    'attribute_3',\n    #'product_code',\n]\nprint(len(cat_cols))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:29.810714Z","iopub.execute_input":"2022-08-13T15:59:29.811675Z","iopub.status.idle":"2022-08-13T15:59:29.817681Z","shell.execute_reply.started":"2022-08-13T15:59:29.811625Z","shell.execute_reply":"2022-08-13T15:59:29.816598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## split by failure","metadata":{}},{"cell_type":"code","source":"nrows, ncols = 2, 3\nfig, ax = plt.subplots(nrows, ncols, figsize=(10, 10))\nax = ax.ravel()\nfor i, var in enumerate(cat_cols):\n    sns.histplot(data=df_train, x=var, hue='failure', ax=ax[i])\n    ax[i].set_label(var)\nax = ax.reshape(nrows, ncols)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:29.819497Z","iopub.execute_input":"2022-08-13T15:59:29.820270Z","iopub.status.idle":"2022-08-13T15:59:30.984363Z","shell.execute_reply.started":"2022-08-13T15:59:29.820217Z","shell.execute_reply":"2022-08-13T15:59:30.983232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## split by product code","metadata":{}},{"cell_type":"code","source":"nrows, ncols = 2, 2\nfig, ax = plt.subplots(nrows, ncols, figsize=(10, 10))\nax = ax.ravel()\nfor i, var in enumerate(cat_cols):\n    sns.histplot(data=df_train, x=var, hue='product_code', ax=ax[i], fill=False)\n    ax[i].set_label(var)\nax = ax.reshape(nrows, ncols)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:30.985935Z","iopub.execute_input":"2022-08-13T15:59:30.987642Z","iopub.status.idle":"2022-08-13T15:59:32.566043Z","shell.execute_reply.started":"2022-08-13T15:59:30.987591Z","shell.execute_reply":"2022-08-13T15:59:32.564813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation matrix","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(20, 20))\nsns.heatmap(df_train.corr(), annot=True, ax=ax, cmap=sns.color_palette(\"coolwarm\", as_cmap=True))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:32.567828Z","iopub.execute_input":"2022-08-13T15:59:32.568304Z","iopub.status.idle":"2022-08-13T15:59:35.191112Z","shell.execute_reply.started":"2022-08-13T15:59:32.568256Z","shell.execute_reply":"2022-08-13T15:59:35.190222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlations of material attributes","metadata":{}},{"cell_type":"code","source":"# not sure how valid this is, OrdinalEncoding implies an ordering to attributes\n\n# create unique list of attributes 0 and 1\nattributes = [f'attribute_{v}' for v in range(4)]\nunique_attributes = df_train['attribute_0'].unique().tolist()\nunique_attributes.extend(df_train['attribute_1'].unique().tolist())\nunique_attributes = list(set(unique_attributes))\n\nattribute_mapper = {k: i for i, k in enumerate(unique_attributes)}\n\nview = df_train[attributes]\nview.loc[:, 'attribute_0'] = view['attribute_0'].map(attribute_mapper)\nview.loc[:, 'attribute_1'] = view['attribute_1'].map(attribute_mapper)\n\nfig, ax = plt.subplots()\nsns.heatmap(view.corr(), annot=True, ax=ax, cmap=sns.color_palette(\"coolwarm\", as_cmap=True))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:35.192415Z","iopub.execute_input":"2022-08-13T15:59:35.192969Z","iopub.status.idle":"2022-08-13T15:59:35.500014Z","shell.execute_reply.started":"2022-08-13T15:59:35.192934Z","shell.execute_reply":"2022-08-13T15:59:35.498797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Failure rate by product code","metadata":{}},{"cell_type":"code","source":"view = df_train.groupby(['product_code', 'failure']).size().unstack()\nview['failure_rate'] = view[1] / (view[0] + view[1])\nprint(view)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:35.501098Z","iopub.execute_input":"2022-08-13T15:59:35.501782Z","iopub.status.idle":"2022-08-13T15:59:35.518618Z","shell.execute_reply.started":"2022-08-13T15:59:35.501743Z","shell.execute_reply":"2022-08-13T15:59:35.517233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dealing with missing values\n* For a baseline model, filling with mean seems a reasonable start\n    * Distributions seem approximately Gaussian (to be checked)\n* measurement_17 has \"some\" correlation with other variables, could be exploited during better imputation methods","metadata":{}},{"cell_type":"markdown","source":"# Class imbalance","metadata":{}},{"cell_type":"code","source":"df_train[df_train['failure'] == 0].shape[0] / df_train[df_train['failure'] == 1].shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:35.520223Z","iopub.execute_input":"2022-08-13T15:59:35.520697Z","iopub.status.idle":"2022-08-13T15:59:35.536563Z","shell.execute_reply.started":"2022-08-13T15:59:35.520653Z","shell.execute_reply":"2022-08-13T15:59:35.535483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline logistic regression model","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(['failure', 'product_code'], axis='columns')\ny = df_train['failure']\n\ncat_cols = [v for v in X.columns if X[v].dtype in ['object', 'int' ]]\nnumerical_cols = [v for v in X.columns if v not in cat_cols]\n\nint_cols = [\n    # 'attribute_0',\n    # 'attribute_1',\n    'attribute_2',\n    'attribute_3',\n]\n\nprint(cat_cols)\nprint(numerical_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:35.537974Z","iopub.execute_input":"2022-08-13T15:59:35.538292Z","iopub.status.idle":"2022-08-13T15:59:35.549534Z","shell.execute_reply.started":"2022-08-13T15:59:35.538263Z","shell.execute_reply":"2022-08-13T15:59:35.548160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder, LabelBinarizer, StandardScaler\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import LogisticRegression\n\n# impute missing values\nimputer = Pipeline(\n    [\n        ('mean_imputer', SimpleImputer(add_indicator=False))\n        # ('mean_imputer', SimpleImputer(add_indicator=True))\n        # ('knn_imputer', KNNImputer(add_indicator=False))\n        # ('knn_imputer', KNNImputer(add_indicator=True))\n    ],\n)\n\n# OneHotEncode categorical features\ncategory_transformer = Pipeline(\n    [\n        ('OneHotEncoder', OneHotEncoder(handle_unknown='ignore', sparse=False)),\n    ],\n)\n\n# total preprocessing\npreproc = ColumnTransformer(\n    transformers = [\n        ('imputer', imputer, numerical_cols),\n        ('OneHotEncoder', category_transformer, cat_cols),\n        # ('InteractionTerms', PolynomialFeatures(interaction_only=True), int_cols),\n        #('PCA', PCA(n_components=5), measurements),\n    ], remainder='passthrough'\n)\n\n# FIXME class weight balanced?\n# penalty = 'l2'\npenalty = 'l1'\n# LR_model = LogisticRegression(class_weight=None, max_iter=10000, penalty=penalty, solver='liblinear')\nLR_model = LogisticRegression(class_weight='balanced', max_iter=10000, penalty=penalty, solver='liblinear')\n\n# final model\npipe = Pipeline(\n    [\n        ('Preprocessing', preproc),\n        ('Scaler', StandardScaler()),\n        ('LogisticRegression', LR_model),\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:35.551474Z","iopub.execute_input":"2022-08-13T15:59:35.552484Z","iopub.status.idle":"2022-08-13T15:59:35.571048Z","shell.execute_reply.started":"2022-08-13T15:59:35.552435Z","shell.execute_reply":"2022-08-13T15:59:35.570092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_validate\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold\nfrom sklearn.metrics import roc_auc_score\n\nscoring = ['roc_auc']\nscores = list()\n\n# cv = StratifiedKFold(6, shuffle=True, random_state=42)\n# cv_results = cross_validate(pipe, X, y, cv=cv, scoring=scoring)\n\ngroups = df_train['product_code']\ncv = GroupKFold(len(groups.unique()))\ncv_results = cross_validate(pipe, X, y, cv=cv, scoring=scoring, groups=groups)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:59:35.573246Z","iopub.execute_input":"2022-08-13T15:59:35.574134Z","iopub.status.idle":"2022-08-13T16:00:24.288164Z","shell.execute_reply.started":"2022-08-13T15:59:35.574097Z","shell.execute_reply":"2022-08-13T16:00:24.286435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(cv_results['test_roc_auc'].mean(), cv_results['test_roc_auc'].std())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:00:24.290524Z","iopub.execute_input":"2022-08-13T16:00:24.291284Z","iopub.status.idle":"2022-08-13T16:00:24.311043Z","shell.execute_reply.started":"2022-08-13T16:00:24.291221Z","shell.execute_reply":"2022-08-13T16:00:24.309495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"pipe.fit(X, y)\npred = pipe.predict_proba(df_test)\npred","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:00:24.313607Z","iopub.execute_input":"2022-08-13T16:00:24.321799Z","iopub.status.idle":"2022-08-13T16:00:38.743333Z","shell.execute_reply.started":"2022-08-13T16:00:24.321704Z","shell.execute_reply":"2022-08-13T16:00:38.741796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv')\nsubmission['failure'] = pred[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:00:38.753597Z","iopub.execute_input":"2022-08-13T16:00:38.754912Z","iopub.status.idle":"2022-08-13T16:00:38.787222Z","shell.execute_reply.started":"2022-08-13T16:00:38.754838Z","shell.execute_reply":"2022-08-13T16:00:38.785611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:00:38.794982Z","iopub.execute_input":"2022-08-13T16:00:38.799129Z","iopub.status.idle":"2022-08-13T16:00:38.879788Z","shell.execute_reply.started":"2022-08-13T16:00:38.799045Z","shell.execute_reply":"2022-08-13T16:00:38.878930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:00:38.882937Z","iopub.execute_input":"2022-08-13T16:00:38.883394Z","iopub.status.idle":"2022-08-13T16:00:38.896239Z","shell.execute_reply.started":"2022-08-13T16:00:38.883346Z","shell.execute_reply":"2022-08-13T16:00:38.895492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}