{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T02:15:06.338047Z","iopub.execute_input":"2022-08-12T02:15:06.338517Z","iopub.status.idle":"2022-08-12T02:15:06.367242Z","shell.execute_reply.started":"2022-08-12T02:15:06.338426Z","shell.execute_reply":"2022-08-12T02:15:06.366361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.Read CSV","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv',index_col='id')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv',index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:06.368784Z","iopub.execute_input":"2022-08-12T02:15:06.369307Z","iopub.status.idle":"2022-08-12T02:15:06.639651Z","shell.execute_reply.started":"2022-08-12T02:15:06.369275Z","shell.execute_reply":"2022-08-12T02:15:06.638485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_num = len(train)\ntest_num = len(test)\n\nall_data = pd.concat([train,test],axis=0)\nall_data['attribute_0'] = all_data['attribute_0'].str.replace('material_','').astype('int')\nall_data['attribute_1'] = all_data['attribute_1'].str.replace('material_','').astype('int')\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:06.640741Z","iopub.execute_input":"2022-08-12T02:15:06.641076Z","iopub.status.idle":"2022-08-12T02:15:06.803485Z","shell.execute_reply.started":"2022-08-12T02:15:06.641046Z","shell.execute_reply":"2022-08-12T02:15:06.802265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.EDA","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/ohba0321/eda-lightgbm-cross-validation?kernelSessionId=102956701","metadata":{}},{"cell_type":"code","source":"# train.head(10)\n\n# test.head(10)\n\n# train.info()\n\n# train.describe(include='all')\n\n# test.info()\n\n# test.describe(include='all')\n\n# all_data['attribute_0'] = all_data['attribute_0'].str.replace('material_','').astype('int')\n# all_data['attribute_1'] = all_data['attribute_1'].str.replace('material_','').astype('int')\n\n# all_data.pivot_table(index=['product_code','attribute_0','attribute_1','attribute_2','attribute_3'],values='failure',aggfunc=['count','mean'])\n\n# memo:\n\n# - Train data and test data has missing values.\n# - Train data and test data can be classified by 'product_code','attribute_0','attribute_1','attribute_2' and 'attribute_3'.This may be a clue to supplement missing values.\n\n# import matplotlib.pyplot as plt\n\n# train_corr = train.corr()\n# fig,ax = plt.subplots(figsize=(12,12))\n# im = ax.imshow(train_corr,cmap='coolwarm')\n# ax.set_xticks(range(len(train_corr.index)))\n# ax.set_yticks(range(len(train_corr.index)))\n# ax.set_xticklabels(train_corr.index,rotation=90)\n# ax.set_yticklabels(train_corr.index,rotation=0)\n# plt.colorbar(im,shrink=0.8);\n\n# test_corr = test.corr()\n# fig,ax = plt.subplots(figsize=(12,12))\n# im = ax.imshow(train_corr,cmap='coolwarm')\n# ax.set_xticks(range(len(train_corr.index)))\n# ax.set_yticks(range(len(train_corr.index)))\n# ax.set_xticklabels(train_corr.index,rotation=90)\n# ax.set_yticklabels(train_corr.index,rotation=0)\n# plt.colorbar(im,shrink=0.8);\n\n# import seaborn as sns\n\n# def hist_and_bar(df,hue=None,bins=10):\n#     # df:DataFrame\n#     len_col = len(df.columns)\n#     ax_col = np.ceil(np.sqrt(len_col)).astype(int)\n#     ax_row = len_col // ax_col\n#     if ax_row * ax_col < len_col:\n#         ax_row += 1\n    \n#     fig,ax = plt.subplots(ax_row, ax_col, figsize=(24,24))\n#     for i, col in enumerate(df.columns):\n#         if df[col].dtypes !='object':\n#             sns.histplot(df,x=col,hue=hue,multiple='stack',bins=bins,ax=ax[i//ax_col, i%ax_col])\n#             ax[i//ax_col, i%ax_col].set_title(col)\n#         else:\n#             sns.histplot(df,x=col,hue=hue,multiple='stack',ax=ax[i//ax_col, i%ax_col])\n#             ax[i//ax_col, i%ax_col].tick_params(axis='x',rotation=0)\n#             ax[i//ax_col, i%ax_col].set_title(col)\n\n# hist_and_bar(train,hue='failure',bins=50)\n\n# hist_and_bar(test,bins=50)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:06.805735Z","iopub.execute_input":"2022-08-12T02:15:06.806038Z","iopub.status.idle":"2022-08-12T02:15:06.811919Z","shell.execute_reply.started":"2022-08-12T02:15:06.806009Z","shell.execute_reply":"2022-08-12T02:15:06.811013Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Completing missing values","metadata":{}},{"cell_type":"code","source":"from sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:06.813749Z","iopub.execute_input":"2022-08-12T02:15:06.814320Z","iopub.status.idle":"2022-08-12T02:15:07.439699Z","shell.execute_reply.started":"2022-08-12T02:15:06.814278Z","shell.execute_reply":"2022-08-12T02:15:07.438641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data_withoutcode = all_data.drop('product_code',axis=1)\niterative_imputer = IterativeImputer(max_iter=100000,tol=0.0001)\nall_data_withoutcode = pd.DataFrame(iterative_imputer.fit_transform(all_data_withoutcode),index=all_data_withoutcode.index,columns=all_data_withoutcode.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:07.440862Z","iopub.execute_input":"2022-08-12T02:15:07.441155Z","iopub.status.idle":"2022-08-12T02:15:22.434392Z","shell.execute_reply.started":"2022-08-12T02:15:07.441128Z","shell.execute_reply":"2022-08-12T02:15:22.432999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data_withoutcode.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.440874Z","iopub.execute_input":"2022-08-12T02:15:22.444022Z","iopub.status.idle":"2022-08-12T02:15:22.476862Z","shell.execute_reply.started":"2022-08-12T02:15:22.443970Z","shell.execute_reply":"2022-08-12T02:15:22.475574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.Prediction","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# from sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.478754Z","iopub.execute_input":"2022-08-12T02:15:22.479535Z","iopub.status.idle":"2022-08-12T02:15:22.484725Z","shell.execute_reply.started":"2022-08-12T02:15:22.479486Z","shell.execute_reply":"2022-08-12T02:15:22.483483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import classification_report\n# from sklearn.metrics import confusion_matrix\n# from sklearn.metrics import roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.488581Z","iopub.execute_input":"2022-08-12T02:15:22.489174Z","iopub.status.idle":"2022-08-12T02:15:22.494875Z","shell.execute_reply.started":"2022-08-12T02:15:22.489139Z","shell.execute_reply":"2022-08-12T02:15:22.494018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from optuna.integration import lightgbm\n# import lightgbm","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.496597Z","iopub.execute_input":"2022-08-12T02:15:22.497231Z","iopub.status.idle":"2022-08-12T02:15:22.504305Z","shell.execute_reply.started":"2022-08-12T02:15:22.497165Z","shell.execute_reply":"2022-08-12T02:15:22.503423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for idx1 in range(4):\n#     for idx2 in range(idx1+1,4):\n#         all_data_withoutcode['attribute_' + str(idx1) +'*' + str(idx2)] = all_data_withoutcode['attribute_' + str(idx1)] * all_data_withoutcode['attribute_' + str(idx2)]\n\n# for idx1 in range(18):\n#     for idx2 in range(idx1+1,18):\n#         all_data_withoutcode['measurement_' + str(idx1) +'*' + str(idx2)] = all_data_withoutcode['measurement_' + str(idx1)] * X_train['measurement_' + str(idx2)]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.506035Z","iopub.execute_input":"2022-08-12T02:15:22.506465Z","iopub.status.idle":"2022-08-12T02:15:22.520316Z","shell.execute_reply.started":"2022-08-12T02:15:22.506416Z","shell.execute_reply":"2022-08-12T02:15:22.519052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imp = all_data_withoutcode[0:train_num]\nX_train = train_imp.drop('failure',axis=1)\ny_train = train_imp['failure']\ntest_imp = all_data_withoutcode.iloc[train_num:]\nX_test = test_imp.drop('failure',axis=1)\ny_test = test_imp['failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.521789Z","iopub.execute_input":"2022-08-12T02:15:22.523474Z","iopub.status.idle":"2022-08-12T02:15:22.535944Z","shell.execute_reply.started":"2022-08-12T02:15:22.523428Z","shell.execute_reply":"2022-08-12T02:15:22.535061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.preprocessing import StandardScaler\n# ss = StandardScaler()\n# X_train = pd.DataFrame(ss.fit_transform(X_train),index=X_train.index,columns=X_train.columns)\n# X_test = pd.DataFrame(ss.transform(X_test),index=X_test.index,columns=X_test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.536991Z","iopub.execute_input":"2022-08-12T02:15:22.537854Z","iopub.status.idle":"2022-08-12T02:15:22.545436Z","shell.execute_reply.started":"2022-08-12T02:15:22.537822Z","shell.execute_reply":"2022-08-12T02:15:22.544667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# kf=KFold(n_splits=5,shuffle=True,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.546572Z","iopub.execute_input":"2022-08-12T02:15:22.547031Z","iopub.status.idle":"2022-08-12T02:15:22.555213Z","shell.execute_reply.started":"2022-08-12T02:15:22.547002Z","shell.execute_reply":"2022-08-12T02:15:22.554455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# models = []\n# for i,(tr_num,val_num) in enumerate(kf.split(X_train,y_train)):\n#     print(f'=====round({i})=====')\n#     # X_tr,X_val,y_tr,y_val = train_test_split(X_train,y_train,test_size=0.25)\n#     X_tr = X_train.iloc[tr_num]\n#     y_tr = y_train.iloc[tr_num]\n#     X_val = X_train.iloc[val_num]\n#     y_val = y_train.iloc[val_num]\n#     train_set = lightgbm.Dataset(X_tr,label=y_tr)\n#     val_set = lightgbm.Dataset(X_val,label=y_val)\n#     callbacks =[\n#                     lightgbm.early_stopping(stopping_rounds=100,first_metric_only=True,verbose=0),\n#                     lightgbm.log_evaluation(1000)\n\n#                 ]\n#     params = {\n#                 'objective' : 'binary',\n#                 'boosting' : 'gbdt',\n#                 'metrics' : ['binary_logloss','auc'],\n#                 'first_metric_only' : True,\n#                 'learning_rate' : 0.0001,\n#                 # 'num_leaves':20,\n#                 # 'max_depth':5,\n#                 # 'min_data_in_leaf':1,\n#                 # 'is_unbalance':True,\n#                 # 'bagging_fraction':0.9,\n#                 # 'bagging_freq':5,\n#                 'verbosity': -1,\n#                 # 'feature_fraction':0.7,\n#                 'min_gain_to_split':0.1,\n#               }\n#     lgb = lightgbm.train(params=params,\n#                          train_set=train_set,\n#                          num_boost_round=100000,\n#                          valid_sets=[train_set,val_set],\n#                          valid_names=['train','val'],\n#                          # categorical_feature=categorical_feature,\n#                          callbacks=callbacks,\n#                          verbose_eval='warn'\n#                         )\n#     print(f'round({i}) best score=======>')\n#     print(lgb.best_score)\n#     models.append(lgb)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.556399Z","iopub.execute_input":"2022-08-12T02:15:22.556871Z","iopub.status.idle":"2022-08-12T02:15:22.565156Z","shell.execute_reply.started":"2022-08-12T02:15:22.556842Z","shell.execute_reply":"2022-08-12T02:15:22.564431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_test_pred = pd.DataFrame(index = test.index)\n# for i,lgb in enumerate(models):\n#     test_pred = lgb.predict(X_test)\n#     y_test_pred['pred_' + str(i)] = pd.Series(test_pred,index=X_test.index)\n# submission = y_test_pred.mean(axis=1)\nsubmission = y_test\nsubmission = submission.reset_index()\n# submission.columns = ['id','failure']\nsubmission.to_csv('../working/submission.csv',index=False,header=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.566106Z","iopub.execute_input":"2022-08-12T02:15:22.566638Z","iopub.status.idle":"2022-08-12T02:15:22.633377Z","shell.execute_reply.started":"2022-08-12T02:15:22.566609Z","shell.execute_reply":"2022-08-12T02:15:22.632394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importances = pd.DataFrame(index=X_train.columns)\n# for i,lgb in enumerate(models):\n#     importances['model_'+str(i)]=pd.Series(lgb.feature_importance('gain'),index=X_train.columns)\n\n# importances['sum']=importances.sum(axis=1)\n\n# importances.to_csv('../working/importances.csv')\n# importances.sort_values('sum',ascending=False).head(20)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.635345Z","iopub.execute_input":"2022-08-12T02:15:22.635727Z","iopub.status.idle":"2022-08-12T02:15:22.639715Z","shell.execute_reply.started":"2022-08-12T02:15:22.635642Z","shell.execute_reply":"2022-08-12T02:15:22.638945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:22.640673Z","iopub.execute_input":"2022-08-12T02:15:22.641421Z","iopub.status.idle":"2022-08-12T02:15:22.653038Z","shell.execute_reply.started":"2022-08-12T02:15:22.641346Z","shell.execute_reply":"2022-08-12T02:15:22.652232Z"},"trusted":true},"execution_count":null,"outputs":[]}]}