{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. Preparation","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom itertools import cycle, islice\nimport matplotlib.pylab as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedShuffleSplit\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import roc_auc_score\n\nplt.style.use(\"ggplot\")\ncolor_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]\ncolor_cycle = cycle(plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T17:55:45.939698Z","iopub.execute_input":"2022-08-01T17:55:45.940090Z","iopub.status.idle":"2022-08-01T17:55:45.947301Z","shell.execute_reply.started":"2022-08-01T17:55:45.940057Z","shell.execute_reply":"2022-08-01T17:55:45.946333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\",index_col='id')\ntest_df = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv',index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:46.635467Z","iopub.execute_input":"2022-08-01T17:55:46.636664Z","iopub.status.idle":"2022-08-01T17:55:46.813666Z","shell.execute_reply.started":"2022-08-01T17:55:46.636622Z","shell.execute_reply":"2022-08-01T17:55:46.812289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Exploring Data","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:47.376037Z","iopub.execute_input":"2022-08-01T17:55:47.376654Z","iopub.status.idle":"2022-08-01T17:55:47.411770Z","shell.execute_reply.started":"2022-08-01T17:55:47.376620Z","shell.execute_reply":"2022-08-01T17:55:47.410545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Product Code\n\n* Test set has totally different product codes from train dataset. It might be a good idea to predict a similar model type from trainingset.\n\n*  Product C has slightly more samples than the others but generally they\nre balanced. Failure ratios about 21%.","metadata":{}},{"cell_type":"code","source":"train_df.product_code.value_counts(), test_df.product_code.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:47.777287Z","iopub.execute_input":"2022-08-01T17:55:47.778140Z","iopub.status.idle":"2022-08-01T17:55:47.788143Z","shell.execute_reply.started":"2022-08-01T17:55:47.778109Z","shell.execute_reply":"2022-08-01T17:55:47.787348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1,2, figsize=(12,5),sharey=True)\n\ntrain_df.groupby('product_code')['loading'].count().plot(kind='bar', ax=axs[0],title=\"Train Product Type\",color=color_pal);\ntest_df.groupby('product_code')['loading'].count().plot(kind='bar', ax=axs[1],title=\"Test Procut Type\",color=color_pal[5:] +color_pal);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T17:55:48.251137Z","iopub.execute_input":"2022-08-01T17:55:48.254090Z","iopub.status.idle":"2022-08-01T17:55:48.553888Z","shell.execute_reply.started":"2022-08-01T17:55:48.254042Z","shell.execute_reply":"2022-08-01T17:55:48.552776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making color cycle\ncolors = list(islice(cycle(['b','r']),None, len(train_df.product_code.unique()) *2))\n\ntrain_df.groupby(['product_code','failure'])['loading'].count().plot(kind='bar',color=colors,figsize=(12, 5),title= 'Failure Ratio by Product Category');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-01T17:55:48.939712Z","iopub.execute_input":"2022-08-01T17:55:48.940105Z","iopub.status.idle":"2022-08-01T17:55:49.110349Z","shell.execute_reply.started":"2022-08-01T17:55:48.940073Z","shell.execute_reply":"2022-08-01T17:55:49.109120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate ratio of failures\ntrain_df.groupby('product_code')['failure'].value_counts(normalize=True).mul(100)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:49.617057Z","iopub.execute_input":"2022-08-01T17:55:49.617454Z","iopub.status.idle":"2022-08-01T17:55:49.631015Z","shell.execute_reply.started":"2022-08-01T17:55:49.617421Z","shell.execute_reply":"2022-08-01T17:55:49.629746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading \n\n#### Loading shows absorbed amount of fluid. It seems very important to predict failure ratio.\n\n* C succeeded absorbing up to 374.33 liquid. That is probably the reason why C was tested more than the others.\n\n* H was tested like C according to the maximum loading data but the number of samples are least in the test data product categories.","metadata":{}},{"cell_type":"code","source":"ax =train_df[train_df['failure']==0].groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']}).plot(kind='bar',figsize=(12, 7), title=\"Success Data by Product Type\");\nax.legend(loc='upper center', bbox_to_anchor=(0.5, -0.10),\n          fancybox=True, shadow=True, ncol=5);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:50.023619Z","iopub.execute_input":"2022-08-01T17:55:50.024677Z","iopub.status.idle":"2022-08-01T17:55:50.232584Z","shell.execute_reply.started":"2022-08-01T17:55:50.024639Z","shell.execute_reply":"2022-08-01T17:55:50.231342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax =train_df[train_df['failure']==1].groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']}).plot(kind='bar',figsize=(12, 7), title=\"Failure Data by Product\");\nax.legend(loc='upper center', bbox_to_anchor=(0.5, -0.10),\n          fancybox=True, shadow=True, ncol=5);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:50.350020Z","iopub.execute_input":"2022-08-01T17:55:50.351171Z","iopub.status.idle":"2022-08-01T17:55:50.596404Z","shell.execute_reply.started":"2022-08-01T17:55:50.351131Z","shell.execute_reply":"2022-08-01T17:55:50.595313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df['failure']==0].groupby(['product_code','failure']).agg({'loading': ['mean', 'min', 'max']})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:50.943034Z","iopub.execute_input":"2022-08-01T17:55:50.943416Z","iopub.status.idle":"2022-08-01T17:55:50.965447Z","shell.execute_reply.started":"2022-08-01T17:55:50.943384Z","shell.execute_reply":"2022-08-01T17:55:50.964226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df['failure']==1].groupby(['product_code','failure']).agg({'loading': ['mean', 'min', 'max']})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:51.471161Z","iopub.execute_input":"2022-08-01T17:55:51.471564Z","iopub.status.idle":"2022-08-01T17:55:51.491430Z","shell.execute_reply.started":"2022-08-01T17:55:51.471533Z","shell.execute_reply":"2022-08-01T17:55:51.490492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.groupby('product_code').agg({'loading': ['mean', 'min', 'max']})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:51.809781Z","iopub.execute_input":"2022-08-01T17:55:51.810176Z","iopub.status.idle":"2022-08-01T17:55:51.829246Z","shell.execute_reply.started":"2022-08-01T17:55:51.810144Z","shell.execute_reply":"2022-08-01T17:55:51.827949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax =test_df.groupby(['product_code']).agg({'loading': ['mean', 'min', 'max']}).plot(kind='bar',figsize=(12, 7), title=\"Failure Data by Product\");\nax.legend(loc='upper center', bbox_to_anchor=(0.5, -0.10),\n          fancybox=True, shadow=True, ncol=5);","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:52.144231Z","iopub.execute_input":"2022-08-01T17:55:52.145069Z","iopub.status.idle":"2022-08-01T17:55:52.383242Z","shell.execute_reply.started":"2022-08-01T17:55:52.145031Z","shell.execute_reply":"2022-08-01T17:55:52.381892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading data have some missing ","metadata":{}},{"cell_type":"code","source":"train_df[train_df['loading'].isna()].groupby(['product_code','failure'])['attribute_0'].count()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:52.554865Z","iopub.execute_input":"2022-08-01T17:55:52.555281Z","iopub.status.idle":"2022-08-01T17:55:52.567485Z","shell.execute_reply.started":"2022-08-01T17:55:52.555230Z","shell.execute_reply":"2022-08-01T17:55:52.566291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" test_df[test_df['loading'].isna()].groupby(['product_code'])['attribute_0'].count()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:52.919867Z","iopub.execute_input":"2022-08-01T17:55:52.921020Z","iopub.status.idle":"2022-08-01T17:55:52.931883Z","shell.execute_reply.started":"2022-08-01T17:55:52.920976Z","shell.execute_reply":"2022-08-01T17:55:52.931015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Attributes 0-3\n\n#### There are four columns about attributes. The first two are about materials and the last two are just categorical numbers. \n\n* Looks like Product code was difined by the Attributes combinations. Each product code has only one attribute combinations.","metadata":{}},{"cell_type":"code","source":"train_df.groupby(['product_code','attribute_0','attribute_1','attribute_2','attribute_3']).size()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:53.313835Z","iopub.execute_input":"2022-08-01T17:55:53.314213Z","iopub.status.idle":"2022-08-01T17:55:53.333665Z","shell.execute_reply.started":"2022-08-01T17:55:53.314185Z","shell.execute_reply":"2022-08-01T17:55:53.332636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.groupby(['product_code','attribute_0','attribute_1','attribute_2','attribute_3']).size()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:53.735802Z","iopub.execute_input":"2022-08-01T17:55:53.736193Z","iopub.status.idle":"2022-08-01T17:55:53.755145Z","shell.execute_reply.started":"2022-08-01T17:55:53.736159Z","shell.execute_reply":"2022-08-01T17:55:53.753970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Mesurement 0-17\n\n#### There are 18 columns about mesurements. Mesurements except for the first two mesurement have some missing values. \n\n* Mesurement 0-2 seems essential for testing and some of meaurements after them are skipped for some conditions. Latter mesurements are skipped more.\n","metadata":{}},{"cell_type":"code","source":"measurements = [a for a in train_df.columns if a.startswith('measurement')]\ntrain_df[measurements].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:54.136937Z","iopub.execute_input":"2022-08-01T17:55:54.138037Z","iopub.status.idle":"2022-08-01T17:55:54.150164Z","shell.execute_reply.started":"2022-08-01T17:55:54.137957Z","shell.execute_reply":"2022-08-01T17:55:54.148891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_plot(df,num,name):\n    a = train_df[measurements].notna().sum().plot.bar(title=f'Mesurement Missing Values Distribution {name}',ax=axs[num])\n    MAX = max(train_df[measurements].notna().sum())\n    MIN = min(train_df[measurements].notna().sum())\n    a.set_ybound(MIN-300,MAX+300)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:54.512445Z","iopub.execute_input":"2022-08-01T17:55:54.513005Z","iopub.status.idle":"2022-08-01T17:55:54.518109Z","shell.execute_reply.started":"2022-08-01T17:55:54.512974Z","shell.execute_reply":"2022-08-01T17:55:54.517358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1,2, figsize=(12,5),sharey=True)\n\nmissing_plot(train_df,0,'Train')\nmissing_plot(test_df,1,'Test')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:54.704492Z","iopub.execute_input":"2022-08-01T17:55:54.705045Z","iopub.status.idle":"2022-08-01T17:55:55.133277Z","shell.execute_reply.started":"2022-08-01T17:55:54.705016Z","shell.execute_reply":"2022-08-01T17:55:55.131987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isna().sum(axis=1).unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.135118Z","iopub.execute_input":"2022-08-01T17:55:55.135463Z","iopub.status.idle":"2022-08-01T17:55:55.148757Z","shell.execute_reply.started":"2022-08-01T17:55:55.135432Z","shell.execute_reply":"2022-08-01T17:55:55.147458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data with six missing values\ntrain_df[train_df.isna().sum(axis=1)==6]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.150250Z","iopub.execute_input":"2022-08-01T17:55:55.150698Z","iopub.status.idle":"2022-08-01T17:55:55.186230Z","shell.execute_reply.started":"2022-08-01T17:55:55.150666Z","shell.execute_reply":"2022-08-01T17:55:55.185442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data with five missing values\ntrain_df[train_df.isna().sum(axis=1)==5]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.277917Z","iopub.execute_input":"2022-08-01T17:55:55.279243Z","iopub.status.idle":"2022-08-01T17:55:55.322166Z","shell.execute_reply.started":"2022-08-01T17:55:55.279187Z","shell.execute_reply":"2022-08-01T17:55:55.320587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data with four missing values\ntrain_df[train_df.isna().sum(axis=1)==4]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.476160Z","iopub.execute_input":"2022-08-01T17:55:55.476558Z","iopub.status.idle":"2022-08-01T17:55:55.517702Z","shell.execute_reply.started":"2022-08-01T17:55:55.476527Z","shell.execute_reply":"2022-08-01T17:55:55.516465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Cleaning Data\n\n* Clean the data based on what I found on Data Exploration.","metadata":{}},{"cell_type":"markdown","source":"Attributes define product types. For now I remove attributes. There might be better ways to recategorize product from the attibutes information. ","metadata":{}},{"cell_type":"code","source":"# remove attribution\ndef no_attribute(df):\n    return df[[a for a in df.columns if not a.startswith('attribute')]]\n\ntrain_df = no_attribute(train_df)\ntest_df = no_attribute(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.662207Z","iopub.execute_input":"2022-08-01T17:55:55.662772Z","iopub.status.idle":"2022-08-01T17:55:55.671652Z","shell.execute_reply.started":"2022-08-01T17:55:55.662741Z","shell.execute_reply":"2022-08-01T17:55:55.670518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Add Missing Values Counts for Each Row","metadata":{}},{"cell_type":"code","source":"def missing_counts(df):\n    df['missing_count'] = df[measurements].isna().sum(axis=1)\n    return df\nmissing_counts(train_df)\nmissing_counts(test_df)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.812676Z","iopub.execute_input":"2022-08-01T17:55:55.813542Z","iopub.status.idle":"2022-08-01T17:55:55.854038Z","shell.execute_reply.started":"2022-08-01T17:55:55.813494Z","shell.execute_reply":"2022-08-01T17:55:55.852849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Add stats by product_code","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"train_stats = train_df.drop(['failure'],axis=1).groupby('product_code').agg(['min','max','std'])\ntest_stats = test_df.groupby('product_code').agg(['min','max','std'])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:55.975309Z","iopub.execute_input":"2022-08-01T17:55:55.975702Z","iopub.status.idle":"2022-08-01T17:55:56.059281Z","shell.execute_reply.started":"2022-08-01T17:55:55.975670Z","shell.execute_reply":"2022-08-01T17:55:56.058213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_stats","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:56.240184Z","iopub.execute_input":"2022-08-01T17:55:56.240597Z","iopub.status.idle":"2022-08-01T17:55:56.301606Z","shell.execute_reply.started":"2022-08-01T17:55:56.240564Z","shell.execute_reply":"2022-08-01T17:55:56.300713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge stats to the original dataframe\npd.options.display.max_columns = 100\nmerged_train = train_df.merge(train_stats, left_on='product_code',right_on='product_code')\nmerged_test  = test_df.merge(test_stats, left_on='product_code',right_on='product_code')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:56.353214Z","iopub.execute_input":"2022-08-01T17:55:56.353817Z","iopub.status.idle":"2022-08-01T17:55:56.410365Z","shell.execute_reply.started":"2022-08-01T17:55:56.353785Z","shell.execute_reply":"2022-08-01T17:55:56.409090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# referred to https://stackoverflow.com/questions/39658574/how-to-drop-columns-which-have-same-values-in-all-rows-via-pandas-or-spark-dataf\n# renove the coloumn have all the same values\ndef drop_nunique(df):\n    nunique= df.nunique()\n    drop_cols = nunique[nunique== 1].index\n    df =df.drop(drop_cols,axis=1)\n    return df\ntrain = drop_nunique(merged_train)\ntest  = drop_nunique(merged_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:56.467836Z","iopub.execute_input":"2022-08-01T17:55:56.468244Z","iopub.status.idle":"2022-08-01T17:55:56.586548Z","shell.execute_reply.started":"2022-08-01T17:55:56.468210Z","shell.execute_reply":"2022-08-01T17:55:56.584968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the row keeping the ratio of product category and failure rate\n# referred to https://stackoverflow.com/questions/45516424/sklearn-train-test-split-on-pandas-stratify-by-multiple-columns\n\ntrain['cat_failure'] = train['product_code'].astype(str)  + train['failure'].astype(str)\n\ntr, ts = train_test_split(train, test_size=0.2, random_state=0, stratify=train[['cat_failure']])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:56.588230Z","iopub.execute_input":"2022-08-01T17:55:56.588595Z","iopub.status.idle":"2022-08-01T17:55:56.715785Z","shell.execute_reply.started":"2022-08-01T17:55:56.588564Z","shell.execute_reply":"2022-08-01T17:55:56.714449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr = tr.drop(['cat_failure','product_code'],axis=1)\nts = ts.drop(['cat_failure','product_code'],axis=1)\n\ny_train = tr.failure\nX_train = tr.drop('failure',axis=1)\ny_test = ts.failure\nX_test = ts.drop('failure',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:56.717656Z","iopub.execute_input":"2022-08-01T17:55:56.718045Z","iopub.status.idle":"2022-08-01T17:55:56.742194Z","shell.execute_reply.started":"2022-08-01T17:55:56.718014Z","shell.execute_reply":"2022-08-01T17:55:56.741178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"xgb = XGBClassifier()\nxgb.fit(X_train,y_train)\npredictions = xgb.predict_proba(X_test)\n\n# get probabilties of failure\nps = [p[1]for p in predictions]\n\n\nroc_auc_score(y_test,ps)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:55:56.827334Z","iopub.execute_input":"2022-08-01T17:55:56.828362Z","iopub.status.idle":"2022-08-01T17:56:03.656745Z","shell.execute_reply.started":"2022-08-01T17:55:56.828310Z","shell.execute_reply":"2022-08-01T17:56:03.655598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making Submissionfile","metadata":{}},{"cell_type":"code","source":"test = test.drop(['product_code'],axis=1)\npredictions = [p[1] for p in xgb.predict_proba(test)]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:56:03.658438Z","iopub.execute_input":"2022-08-01T17:56:03.659074Z","iopub.status.idle":"2022-08-01T17:56:03.718106Z","shell.execute_reply.started":"2022-08-01T17:56:03.659044Z","shell.execute_reply":"2022-08-01T17:56:03.716871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv',index_col='id')\nsubmission['failure'] = predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:56:03.719547Z","iopub.execute_input":"2022-08-01T17:56:03.719891Z","iopub.status.idle":"2022-08-01T17:56:03.737282Z","shell.execute_reply.started":"2022-08-01T17:56:03.719860Z","shell.execute_reply":"2022-08-01T17:56:03.736141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission_1')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T17:56:03.739615Z","iopub.execute_input":"2022-08-01T17:56:03.740329Z","iopub.status.idle":"2022-08-01T17:56:03.787774Z","shell.execute_reply.started":"2022-08-01T17:56:03.740281Z","shell.execute_reply":"2022-08-01T17:56:03.786856Z"},"trusted":true},"execution_count":null,"outputs":[]}]}