{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport lightgbm as lgbm\nfrom sklearn.metrics import roc_auc_score\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T12:28:01.427290Z","iopub.execute_input":"2022-07-23T12:28:01.427628Z","iopub.status.idle":"2022-07-23T12:28:02.768233Z","shell.execute_reply.started":"2022-07-23T12:28:01.427529Z","shell.execute_reply":"2022-07-23T12:28:02.767324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = np.load('../input/tps-apr-2022-shapelet50025/df_3d_s500-25.npy')\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:07.936364Z","iopub.execute_input":"2022-07-23T12:28:07.936664Z","iopub.status.idle":"2022-07-23T12:28:09.399271Z","shell.execute_reply.started":"2022-07-23T12:28:07.936629Z","shell.execute_reply":"2022-07-23T12:28:09.398443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = np.load('../input/tps-apr-2022-shapelet50025/test_3d_s500-25.npy')\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:13.172161Z","iopub.execute_input":"2022-07-23T12:28:13.172443Z","iopub.status.idle":"2022-07-23T12:28:13.593813Z","shell.execute_reply.started":"2022-07-23T12:28:13.172411Z","shell.execute_reply":"2022-07-23T12:28:13.593016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_label = pd.read_csv('../input/tabular-playground-series-apr-2022/train_labels.csv')\ndf_label.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:15.798653Z","iopub.execute_input":"2022-07-23T12:28:15.798991Z","iopub.status.idle":"2022-07-23T12:28:15.831443Z","shell.execute_reply.started":"2022-07-23T12:28:15.798958Z","shell.execute_reply":"2022-07-23T12:28:15.830663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ravel = np.zeros((25968, 13*25))\nfor i in range(25968):\n    df_ravel[i] = df[i].flatten('F')\ndf_ravel.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:26.138365Z","iopub.execute_input":"2022-07-23T12:28:26.138683Z","iopub.status.idle":"2022-07-23T12:28:26.252169Z","shell.execute_reply.started":"2022-07-23T12:28:26.138643Z","shell.execute_reply":"2022-07-23T12:28:26.251359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ravel = np.zeros((12218, 13*25))\nfor i in range(12218):\n    test_ravel[i] = test[i].flatten('F')\ntest_ravel.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:28.927805Z","iopub.execute_input":"2022-07-23T12:28:28.928204Z","iopub.status.idle":"2022-07-23T12:28:28.983307Z","shell.execute_reply.started":"2022-07-23T12:28:28.928174Z","shell.execute_reply":"2022-07-23T12:28:28.982498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## split data","metadata":{}},{"cell_type":"code","source":"trainid = np.random.choice(25968, int(0.8*25968), replace=False)\nvalidid = [i for i in range(25968) if i not in trainid]\nlen(trainid)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:34.269510Z","iopub.execute_input":"2022-07-23T12:28:34.269838Z","iopub.status.idle":"2022-07-23T12:28:34.565419Z","shell.execute_reply.started":"2022-07-23T12:28:34.269801Z","shell.execute_reply":"2022-07-23T12:28:34.564885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df_ravel[trainid]\nX_valid = df_ravel[validid]\n\ny_train = df_label.loc[trainid, 'state']\ny_valid = df_label.loc[validid, 'state']","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:36.334454Z","iopub.execute_input":"2022-07-23T12:28:36.334876Z","iopub.status.idle":"2022-07-23T12:28:36.379023Z","shell.execute_reply.started":"2022-07-23T12:28:36.334843Z","shell.execute_reply":"2022-07-23T12:28:36.378289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_valid.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:28:39.273416Z","iopub.execute_input":"2022-07-23T12:28:39.273702Z","iopub.status.idle":"2022-07-23T12:28:39.278777Z","shell.execute_reply.started":"2022-07-23T12:28:39.273671Z","shell.execute_reply":"2022-07-23T12:28:39.278203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling: LGBM","metadata":{}},{"cell_type":"code","source":"lr0 = 0.1\nlr = []\nfor i in range(320):\n    if i<150:\n        lr.append(lr0)\n    else:\n        new_lr = lr0*0.98\n        lr.append(new_lr)\n        lr0 = new_lr\nlr_callback = lgbm.callback.reset_parameter(learning_rate=lr)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T12:02:04.733568Z","iopub.execute_input":"2022-04-28T12:02:04.733874Z","iopub.status.idle":"2022-04-28T12:02:04.745333Z","shell.execute_reply.started":"2022-04-28T12:02:04.733838Z","shell.execute_reply":"2022-04-28T12:02:04.744037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = lgbm.LGBMClassifier(n_estimators=2500, num_leaves=35, min_child_samples=150,\n                            reg_lambda=0.1, colsample_bytree=0.95, random_state=8)\nmodel.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], eval_metric='AUC')\nmodel.get_params()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-23T12:37:44.706503Z","iopub.execute_input":"2022-07-23T12:37:44.706814Z","iopub.status.idle":"2022-07-23T12:39:10.815497Z","shell.execute_reply.started":"2022-07-23T12:37:44.706786Z","shell.execute_reply":"2022-07-23T12:39:10.814491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evals_result_['valid_0']['auc'][-10:]","metadata":{"execution":{"iopub.status.busy":"2022-04-28T12:02:14.827428Z","iopub.execute_input":"2022-04-28T12:02:14.829753Z","iopub.status.idle":"2022-04-28T12:02:14.837916Z","shell.execute_reply.started":"2022-04-28T12:02:14.829696Z","shell.execute_reply":"2022-04-28T12:02:14.836667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred = model.predict_proba(X_train)[:,1]\nroc_auc_score(y_train, train_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:32:46.181396Z","iopub.execute_input":"2022-07-23T12:32:46.182185Z","iopub.status.idle":"2022-07-23T12:32:47.668105Z","shell.execute_reply.started":"2022-07-23T12:32:46.182142Z","shell.execute_reply":"2022-07-23T12:32:47.667551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(model.evals_result_['valid_0']['auc'][1000:]);","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:43:58.315973Z","iopub.execute_input":"2022-07-23T12:43:58.316261Z","iopub.status.idle":"2022-07-23T12:43:58.513155Z","shell.execute_reply.started":"2022-07-23T12:43:58.316228Z","shell.execute_reply":"2022-07-23T12:43:58.512387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(model.evals_result_['valid_0']['binary_logloss'][:]);","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:44:05.554022Z","iopub.execute_input":"2022-07-23T12:44:05.554284Z","iopub.status.idle":"2022-07-23T12:44:05.748016Z","shell.execute_reply.started":"2022-07-23T12:44:05.554257Z","shell.execute_reply":"2022-07-23T12:44:05.747029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lgbm.cv(model.get_params(), \n#        train_set=lgbm.Dataset(df_ravel, df_label.state),\n#        nfold=10,\n#        metrics='AUC')","metadata":{"execution":{"iopub.status.busy":"2022-04-28T11:03:40.680447Z","iopub.execute_input":"2022-04-28T11:03:40.680762Z","iopub.status.idle":"2022-04-28T11:05:03.910243Z","shell.execute_reply.started":"2022-04-28T11:03:40.680726Z","shell.execute_reply":"2022-04-28T11:05:03.909452Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"subs = pd.read_csv('../input/tabular-playground-series-apr-2022/sample_submission.csv')\nsubs.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T12:05:12.996631Z","iopub.execute_input":"2022-04-28T12:05:12.996967Z","iopub.status.idle":"2022-04-28T12:05:13.019189Z","shell.execute_reply.started":"2022-04-28T12:05:12.996933Z","shell.execute_reply":"2022-04-28T12:05:13.018520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs['state'] = model.predict_proba(test_ravel, raw_score=True)\nsubs.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T12:10:51.160529Z","iopub.execute_input":"2022-04-28T12:10:51.160849Z","iopub.status.idle":"2022-04-28T12:10:51.290640Z","shell.execute_reply.started":"2022-04-28T12:10:51.160818Z","shell.execute_reply":"2022-04-28T12:10:51.289711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs.to_csv('mysubmission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T12:11:36.432701Z","iopub.execute_input":"2022-04-28T12:11:36.433030Z","iopub.status.idle":"2022-04-28T12:11:36.486833Z","shell.execute_reply.started":"2022-04-28T12:11:36.432998Z","shell.execute_reply":"2022-04-28T12:11:36.485633Z"},"trusted":true},"execution_count":null,"outputs":[]}]}