{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import display\n\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import roc_auc_score, roc_curve\nfrom sklearn.preprocessing import StandardScaler\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:36.203014Z","iopub.execute_input":"2022-08-01T11:28:36.204174Z","iopub.status.idle":"2022-08-01T11:28:36.210249Z","shell.execute_reply.started":"2022-08-01T11:28:36.204101Z","shell.execute_reply":"2022-08-01T11:28:36.209289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:37.632344Z","iopub.execute_input":"2022-08-01T11:28:37.633011Z","iopub.status.idle":"2022-08-01T11:28:37.942419Z","shell.execute_reply.started":"2022-08-01T11:28:37.632973Z","shell.execute_reply":"2022-08-01T11:28:37.941434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['product_code'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:39.350456Z","iopub.execute_input":"2022-08-01T11:28:39.350839Z","iopub.status.idle":"2022-08-01T11:28:39.367330Z","shell.execute_reply.started":"2022-08-01T11:28:39.350810Z","shell.execute_reply":"2022-08-01T11:28:39.366185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a0_labels = {'material_7' : 7,\n            'material_5': 5}\na1_labels = {'material_5': 5,\n            'material_6': 6,\n            'material_7': 7,\n            'material_8': 8}\npc_labels = {'A': 0,\n            'B': 1,\n            'C': 2,\n            'D': 3,\n            'E': 4,\n            'F': 5,\n            'G': 6,\n            'H': 7,\n            'I': 8}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:40.110386Z","iopub.execute_input":"2022-08-01T11:28:40.111002Z","iopub.status.idle":"2022-08-01T11:28:40.117486Z","shell.execute_reply.started":"2022-08-01T11:28:40.110964Z","shell.execute_reply":"2022-08-01T11:28:40.116188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['attribute_0'].replace(a0_labels, inplace = True)\ntest['attribute_0'].replace(a0_labels, inplace = True)\ntrain['attribute_1'].replace(a1_labels, inplace = True)\ntest['attribute_1'].replace(a1_labels, inplace = True)\ntrain['product_code'].replace(pc_labels, inplace = True)\ntest['product_code'].replace(pc_labels, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:41.091817Z","iopub.execute_input":"2022-08-01T11:28:41.092532Z","iopub.status.idle":"2022-08-01T11:28:41.203269Z","shell.execute_reply.started":"2022-08-01T11:28:41.092493Z","shell.execute_reply":"2022-08-01T11:28:41.202350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.fillna(train.mean(axis=0))\ntest = test.fillna(test.mean(axis=0))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:43.138317Z","iopub.execute_input":"2022-08-01T11:28:43.140574Z","iopub.status.idle":"2022-08-01T11:28:43.168365Z","shell.execute_reply.started":"2022-08-01T11:28:43.140539Z","shell.execute_reply":"2022-08-01T11:28:43.167322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat = ['product_code','loading', 'attribute_0', 'attribute_1',\n       'attribute_2', 'attribute_3', 'measurement_0', 'measurement_1',\n       'measurement_2', 'measurement_3', 'measurement_4', 'measurement_5',\n       'measurement_6', 'measurement_7', 'measurement_8', 'measurement_9',\n       'measurement_10', 'measurement_11', 'measurement_12', 'measurement_13',\n       'measurement_14', 'measurement_15', 'measurement_16', 'measurement_17']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:57.681473Z","iopub.execute_input":"2022-08-01T11:28:57.682153Z","iopub.status.idle":"2022-08-01T11:28:57.688190Z","shell.execute_reply.started":"2022-08-01T11:28:57.682101Z","shell.execute_reply":"2022-08-01T11:28:57.687137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(columns = 'id', inplace = True)\ntest.drop(columns = 'id', inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:28:59.668549Z","iopub.execute_input":"2022-08-01T11:28:59.669239Z","iopub.status.idle":"2022-08-01T11:28:59.678103Z","shell.execute_reply.started":"2022-08-01T11:28:59.669206Z","shell.execute_reply":"2022-08-01T11:28:59.676632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\ndef objective(trial):\n    \n    param = {\n        'tree_method':'gpu_hist',  # this parameter means using the GPU when training our model to speedup the training process\n        'lambda': trial.suggest_loguniform('lambda', 1e-3, 10.0),\n        'alpha': trial.suggest_loguniform('alpha', 1e-3, 10.0),\n        'colsample_bytree': trial.suggest_categorical('colsample_bytree', [0.3,0.4,0.5,0.6,0.7,0.8,0.9, 1.0]),\n        'subsample': trial.suggest_categorical('subsample', [0.4,0.5,0.6,0.7,0.8,1.0]),\n        'learning_rate': trial.suggest_categorical('learning_rate', [0.008,0.01,0.012,0.014,0.016,0.018, 0.02]),\n        'n_estimators': 10000,\n        'max_depth': trial.suggest_categorical('max_depth', [5,7,9,11,13,15,17]),\n        'random_state': trial.suggest_categorical('random_state', [2020]),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 300),\n    }\n    model = XGBRegressor(**param)  \n    \n    model.fit(X_tr,y_tr,eval_set=[(X_va,y_va)],early_stopping_rounds=100,verbose=False)\n    \n    preds = model.predict(X_va)\n    \n    rmse = roc_auc_score(y_va, preds)\n    \n    return rmse","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:29:00.714630Z","iopub.execute_input":"2022-08-01T11:29:00.715425Z","iopub.status.idle":"2022-08-01T11:29:01.294780Z","shell.execute_reply.started":"2022-08-01T11:29:00.715370Z","shell.execute_reply":"2022-08-01T11:29:01.293866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor  \nmodel = XGBRegressor()\npreds = []\ndef fit_model(X_tr, y_tr, X_va, y_va, test):\n    scaler = StandardScaler()\n    X_tr = scaler.fit_transform(X_tr)\n    X_va = scaler.fit_transform(X_va)\n    test = scaler.transform(test[feat])\n    study = optuna.create_study(direction='maximize')\n    study.optimize(objective, n_trials=30)\n    print('Number of finished trials:', len(study.trials))\n    print('Best trial:', study.best_trial.params)\n    params = study.best_trial.params\n    model = XGBRegressor(**params)\n    model.fit(X_tr, y_tr,eval_set=[(X_va,y_va)],early_stopping_rounds=100,verbose=False)\n    print(\"AUC: {}\".format(roc_auc_score(y_va, model.predict(X_va))))\n    pred = model.predict(test)\n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:29:02.129377Z","iopub.execute_input":"2022-08-01T11:29:02.130367Z","iopub.status.idle":"2022-08-01T11:29:02.219844Z","shell.execute_reply.started":"2022-08-01T11:29:02.130318Z","shell.execute_reply":"2022-08-01T11:29:02.218971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train[feat], train.failure, train.product_code)):\n    X_tr = train.iloc[idx_tr][test.columns]\n    X_va = train.iloc[idx_va][test.columns]\n    X_te = test.copy()\n    y_tr = train.iloc[idx_tr].failure\n    y_va = train.iloc[idx_va].failure\n    fit_model(X_tr, y_tr, X_va, y_va, test)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:29:03.123392Z","iopub.execute_input":"2022-08-01T11:29:03.125436Z","iopub.status.idle":"2022-08-01T11:30:11.409584Z","shell.execute_reply.started":"2022-08-01T11:29:03.125385Z","shell.execute_reply":"2022-08-01T11:30:11.406365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:11:27.656288Z","iopub.execute_input":"2022-08-01T11:11:27.656683Z","iopub.status.idle":"2022-08-01T11:11:27.729523Z","shell.execute_reply.started":"2022-08-01T11:11:27.656650Z","shell.execute_reply":"2022-08-01T11:11:27.728546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame()\nsub['id'] = test['id']\nsub['pred1'] = preds[0]\nsub['pred2'] = preds[1]\nsub['pred3'] = preds[2]\nsub['pred4'] = preds[3]\nsub['pred5'] = preds[4]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:12:19.089760Z","iopub.execute_input":"2022-08-01T11:12:19.090747Z","iopub.status.idle":"2022-08-01T11:12:19.104956Z","shell.execute_reply.started":"2022-08-01T11:12:19.090701Z","shell.execute_reply":"2022-08-01T11:12:19.103994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:12:43.580181Z","iopub.execute_input":"2022-08-01T11:12:43.582136Z","iopub.status.idle":"2022-08-01T11:12:43.589860Z","shell.execute_reply.started":"2022-08-01T11:12:43.582094Z","shell.execute_reply":"2022-08-01T11:12:43.588902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['failure'] = sub[['pred1', 'pred2', 'pred3', 'pred4', 'pred5']].sum(axis =1)/5","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:13:03.466713Z","iopub.execute_input":"2022-08-01T11:13:03.467689Z","iopub.status.idle":"2022-08-01T11:13:03.476169Z","shell.execute_reply.started":"2022-08-01T11:13:03.467642Z","shell.execute_reply":"2022-08-01T11:13:03.475182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = sub[['id','failure']]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:13:14.896344Z","iopub.execute_input":"2022-08-01T11:13:14.896711Z","iopub.status.idle":"2022-08-01T11:13:14.903446Z","shell.execute_reply.started":"2022-08-01T11:13:14.896679Z","shell.execute_reply":"2022-08-01T11:13:14.902400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('sub_tps_aug22_xgbreg_2.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:13:41.146884Z","iopub.execute_input":"2022-08-01T11:13:41.147264Z","iopub.status.idle":"2022-08-01T11:13:41.205409Z","shell.execute_reply.started":"2022-08-01T11:13:41.147232Z","shell.execute_reply":"2022-08-01T11:13:41.204430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}