{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![overview](https://drive.google.com/uc?export=view&id=1xfXg0jemtUSz5n-BuVdSCsFGqYWDIulS )","metadata":{}},{"cell_type":"markdown","source":"|||\n|:--|:--|\n|**train.csv**             | トレーニングセット、臨床情報の完全な履歴が含まれます  |\n|**test.csv**              | テストセット。ベースライン測定のみが含まれます  |\n|**train /**               | トレーニング患者のベースラインCTスキャンがDICOM形式で含まれています | \n|**test /**                | テスト患者のベースラインCTスキャンがDICOM形式で含まれています  |\n|**sample_submission.csv** | 送信形式を示します  |","metadata":{}},{"cell_type":"markdown","source":"![TurnOff](https://drive.google.com/uc?export=view&id=14iabidS4S0Ur7R5smsHYBLZjTP9M3vWY )\n\nYou cannot use the internet in this competition. Turn it off.\n> このコンペではインターネットを使うことはできません。右下のSettingsからインターネットをOFFにします。","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport glob\nimport re\nimport cv2","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-06-02T02:27:24.845002Z","iopub.execute_input":"2021-06-02T02:27:24.845587Z","iopub.status.idle":"2021-06-02T02:27:25.973057Z","shell.execute_reply.started":"2021-06-02T02:27:24.845535Z","shell.execute_reply":"2021-06-02T02:27:25.972011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking the data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:25.974991Z","iopub.execute_input":"2021-06-02T02:27:25.975335Z","iopub.status.idle":"2021-06-02T02:27:26.024221Z","shell.execute_reply.started":"2021-06-02T02:27:25.975304Z","shell.execute_reply":"2021-06-02T02:27:26.023383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"患者によってデータの量、種類は異なる。","metadata":{}},{"cell_type":"markdown","source":"| | | \n|:---|:---|\n|**Patient** | 患者ごとの一意のID（患者のDICOMフォルダの名前も）|\n|**Weeks** | ベースラインCTの前後の相対的な週数（負の場合がある）| \n|**FVC** | 記録された肺容量（ml）|\n|**Percent** | 患者のFVCを、類似した特徴を持つ人の典型的なFVCのパーセントとして概算する計算フィールド |\n|**Age** | 年齢。 |\n|**Sex** | 性別。　Male/Female  |\n|**SmokingStatus** | 喫煙者かどうか。 Currently smokes (喫煙者) / Ex-smoker (元喫煙者) / Never smoked (非喫煙者)|","metadata":{}},{"cell_type":"markdown","source":"## Display DICOM image\npydicomを使って.dcm画像を表示する","metadata":{}},{"cell_type":"code","source":"import pydicom\n\ndef plot_pixel_array(dataset, figsize=(5,5)):\n    plt.figure(figsize=figsize)\n    plt.imshow(dataset.pixel_array, cmap=plt.cm.bone)\n    plt.show()\n\nfile_path = \"../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/1.dcm\"\ndataset = pydicom.dcmread(file_path)\nplot_pixel_array(dataset)","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:26.026008Z","iopub.execute_input":"2021-06-02T02:27:26.026595Z","iopub.status.idle":"2021-06-02T02:27:26.406577Z","shell.execute_reply.started":"2021-06-02T02:27:26.026544Z","shell.execute_reply":"2021-06-02T02:27:26.403482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"参考:https://qiita.com/fukuit/items/ed163f9b566baf3a6c3f","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"### Pay attention to ID = \"ID00007637202177411956430\"\n ID = \"ID00007637202177411956430\"に注目する","metadata":{}},{"cell_type":"code","source":"def extract_num(s, p, ret=0):\n    search = p.search(s)\n    if search:\n        return int(search.groups()[0])\n    else:\n        return ret","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:26.408065Z","iopub.execute_input":"2021-06-02T02:27:26.408438Z","iopub.status.idle":"2021-06-02T02:27:26.414135Z","shell.execute_reply.started":"2021-06-02T02:27:26.408403Z","shell.execute_reply":"2021-06-02T02:27:26.413278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = []\nID = \"ID00007637202177411956430\"\n\nfor file in glob.glob(\"../input/osic-pulmonary-fibrosis-progression/train/\"+ ID +\"/*.dcm\"):\n    filepath.append(file)\n    \np = re.compile(ID +\"/\"+\"(\\d+)\")\nfilepath = sorted(filepath, key=lambda s: extract_num(s, p, float('inf'))) #画像を数字順にsort","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:26.416807Z","iopub.execute_input":"2021-06-02T02:27:26.417289Z","iopub.status.idle":"2021-06-02T02:27:26.441201Z","shell.execute_reply.started":"2021-06-02T02:27:26.417222Z","shell.execute_reply":"2021-06-02T02:27:26.440180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(16,7))\n\nfor i in range(18):\n    plt.subplot(3, 6, i+1)\n    file_path = filepath[i]\n    dataset = pydicom.dcmread(file_path)\n    plt.imshow(dataset.pixel_array, cmap=plt.cm.bone)\n    plt.title(file_path[77:])\n    plt.tick_params(labelbottom=False,\n                    labelleft=False,\n                    labelright=False,\n                    labeltop=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:26.443888Z","iopub.execute_input":"2021-06-02T02:27:26.444423Z","iopub.status.idle":"2021-06-02T02:27:30.435885Z","shell.execute_reply.started":"2021-06-02T02:27:26.444379Z","shell.execute_reply":"2021-06-02T02:27:30.434821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.loc[train_df.Patient == ID]","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:30.437184Z","iopub.execute_input":"2021-06-02T02:27:30.437495Z","iopub.status.idle":"2021-06-02T02:27:30.460467Z","shell.execute_reply.started":"2021-06-02T02:27:30.437466Z","shell.execute_reply":"2021-06-02T02:27:30.459169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Patient_list = list(train_df.Patient.unique())\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(15, 4))\n\na = 0\nb = 0\nc = 0\n\nfor ID in Patient_list:\n    grp = train_df.loc[train_df.Patient == ID]\n    grp = grp[[\"Weeks\",\"FVC\", \"SmokingStatus\"]]\n\n    if grp.iloc[0, 2] == \"Currently smokes\" and a <= 10:\n        ax1.plot(grp.Weeks, grp.FVC, marker=\"o\", color=\"red\")\n        ax1.set_title(\"Currently smokes\") \n        a = a + 1\n    elif grp.iloc[0, 2] == \"Ex-smoker\" and b <= 10:\n        ax2.plot(grp.Weeks, grp.FVC, marker=\"x\", color=\"green\")\n        ax2.set_title(\"Ex-smoker\") \n        b = b + 1\n    elif grp.iloc[0, 2] == \"Never smoked\" and c <= 10:\n        ax3.plot(grp.Weeks, grp.FVC, marker=\"s\", color=\"blue\")\n        ax3.set_title(\"Never smoked\")\n        c = c + 1\n    else:\n        pass","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:30.462228Z","iopub.execute_input":"2021-06-02T02:27:30.462618Z","iopub.status.idle":"2021-06-02T02:27:31.282449Z","shell.execute_reply.started":"2021-06-02T02:27:30.462585Z","shell.execute_reply":"2021-06-02T02:27:31.281478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create train_X & test_X","metadata":{}},{"cell_type":"markdown","source":"### Complement train data\n訓練データの値を補完する。（線形補間）","metadata":{}},{"cell_type":"code","source":"Week = np.arange(-12, 134)\ntrain_df2 = pd.DataFrame(Week, columns = [\"Weeks\"])\ntrain_df2.insert(1, 'FVC', np.nan)\ntrain_df2.insert(2, 'Percent', np.nan)\ntrain_df2.insert(3, 'Age', np.nan)\ntrain_df2.insert(4, 'Sex', np.nan)\ntrain_df2.insert(5, 'SmokingStatus', np.nan)\n\ntrain_id = train_df.loc[train_df.Patient == Patient_list[1]]\ntrain_id = train_id.reset_index()\n\nfor i, D in enumerate(train_id.Weeks):\n    D = D + 12\n    train_df2.at[D, \"FVC\"] = train_id.FVC[i]\n    train_df2.at[D, \"Percent\"] = train_id.Percent[i]\n\ntrain_df2.loc[:, \"Age\"] = train_id.Age[0]\ntrain_df2.loc[:, \"Sex\"] = train_id.Sex[0]\ntrain_df2.loc[:, \"SmokingStatus\"] = train_id.SmokingStatus[0]\n    \ntrain_df2 = train_df2.interpolate('linear', order=2, limit_direction='both')\ntrain_df2","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.286012Z","iopub.execute_input":"2021-06-02T02:27:31.286329Z","iopub.status.idle":"2021-06-02T02:27:31.340581Z","shell.execute_reply.started":"2021-06-02T02:27:31.286299Z","shell.execute_reply":"2021-06-02T02:27:31.339366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,6))\ngrp = train_df2\n\nplt.xlabel(\"Weeks\")\nplt.ylabel(\"FVC\")\nplt.plot(grp.Weeks, grp.FVC, marker=\"x\")\nplt.plot(train_id.Weeks, train_id.FVC, marker=\"o\", markersize=8)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.342209Z","iopub.execute_input":"2021-06-02T02:27:31.342681Z","iopub.status.idle":"2021-06-02T02:27:31.532649Z","shell.execute_reply.started":"2021-06-02T02:27:31.342633Z","shell.execute_reply":"2021-06-02T02:27:31.531530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,6))\ngrp = train_df2\n\nplt.xlabel(\"Weeks\")\nplt.ylabel(\"Percent\")\nplt.plot(grp.Weeks, grp.Percent, marker=\"^\")\nplt.plot(train_id.Weeks, train_id.Percent, marker=\"o\", markersize=8)","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.534442Z","iopub.execute_input":"2021-06-02T02:27:31.534925Z","iopub.status.idle":"2021-06-02T02:27:31.727288Z","shell.execute_reply.started":"2021-06-02T02:27:31.534878Z","shell.execute_reply":"2021-06-02T02:27:31.726196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create training data","metadata":{}},{"cell_type":"code","source":"Week = np.arange(-12, 134)\n\ndef train_layer (ID_N):\n    train_df2   = pd.DataFrame(Week, columns = [\"Weeks\"])\n    train_df_Y  = pd.DataFrame(Week, columns = [\"Weeks\"])\n    train_df_Y.insert(1, 'FVC', np.nan)\n    train_df2.insert(1, 'Percent', np.nan)\n    train_df2.insert(2, 'Age', np.nan)\n    train_df2.insert(3, 'Sex_Male', 0)\n    train_df2.insert(4, 'Sex_Female', 0)\n    train_df2.insert(5, 'Currently smokes', 0)\n    train_df2.insert(6, 'Ex-smoker', 0)\n    train_df2.insert(7, 'Never smoked', 0)\n\n    train_id = train_df.loc[train_df.Patient == Patient_list[ID_N]]\n    train_id = train_id.reset_index()\n\n    for i, D in enumerate(train_id.Weeks):\n        D = D + 12\n        if D <= 133:\n            train_df_Y.at[D, \"FVC\"] = train_id.FVC[i]\n            train_df2.at[D, \"Percent\"] = train_id.Percent[i]\n\n    train_df2.loc[:, \"Age\"] = train_id.Age[0]\n\n    if train_id.Sex[0] == \"Male\":\n        train_df2.loc[:, \"Sex_Male\"] = 1\n    else:\n        train_df2.loc[:, \"Sex_Female\"] = 1\n    \n    if train_id.SmokingStatus[0] == \"Currently smokes\":\n        train_df2.loc[:, \"Currently smokes\"] = 1\n    elif train_id.SmokingStatus[0] == \"Ex-smoker\":\n        train_df2.loc[:, \"Ex-smoker\"] = 1\n    else:\n        train_df2.loc[:, \"Never smoked\"] = 1\n        \n    train_df2 = train_df2.interpolate('linear', order=2, limit_direction='both')\n    train_df_Y = train_df_Y.interpolate('linear', order=2, limit_direction='both')\n    train_df_Y = train_df_Y.astype('int')\n    train_df_Y = train_df_Y.drop([\"Weeks\"], axis=1)\n    \n    return train_df2, train_df_Y","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.728593Z","iopub.execute_input":"2021-06-02T02:27:31.728892Z","iopub.status.idle":"2021-06-02T02:27:31.747181Z","shell.execute_reply.started":"2021-06-02T02:27:31.728863Z","shell.execute_reply":"2021-06-02T02:27:31.745922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_layer(0)[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.749433Z","iopub.execute_input":"2021-06-02T02:27:31.749964Z","iopub.status.idle":"2021-06-02T02:27:31.796979Z","shell.execute_reply.started":"2021-06-02T02:27:31.749912Z","shell.execute_reply":"2021-06-02T02:27:31.796081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_layer(0)[1]","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.798525Z","iopub.execute_input":"2021-06-02T02:27:31.798930Z","iopub.status.idle":"2021-06-02T02:27:31.828564Z","shell.execute_reply.started":"2021-06-02T02:27:31.798896Z","shell.execute_reply":"2021-06-02T02:27:31.827779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_layer(0)[0].to_numpy()\nY_train = train_layer(0)[1].to_numpy()\nsums= 0\n\nfor i in range(1, len(Patient_list)):\n    a = train_layer(i)[0].to_numpy()\n    X_train = np.append(X_train, a, axis=0)\n    \n    b = train_layer(i)[1].to_numpy()\n    Y_train = np.append(Y_train, b)    ","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:31.829500Z","iopub.execute_input":"2021-06-02T02:27:31.829772Z","iopub.status.idle":"2021-06-02T02:27:36.188685Z","shell.execute_reply.started":"2021-06-02T02:27:31.829744Z","shell.execute_reply":"2021-06-02T02:27:36.187508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.190009Z","iopub.execute_input":"2021-06-02T02:27:36.190328Z","iopub.status.idle":"2021-06-02T02:27:36.196213Z","shell.execute_reply.started":"2021-06-02T02:27:36.190298Z","shell.execute_reply":"2021-06-02T02:27:36.195066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.197533Z","iopub.execute_input":"2021-06-02T02:27:36.197819Z","iopub.status.idle":"2021-06-02T02:27:36.212570Z","shell.execute_reply.started":"2021-06-02T02:27:36.197790Z","shell.execute_reply":"2021-06-02T02:27:36.211316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create test data","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\ntest_df","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.214227Z","iopub.execute_input":"2021-06-02T02:27:36.214594Z","iopub.status.idle":"2021-06-02T02:27:36.245330Z","shell.execute_reply.started":"2021-06-02T02:27:36.214560Z","shell.execute_reply":"2021-06-02T02:27:36.244377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Week = np.arange(-12, 134)\nPatient_list_test = list(test_df.Patient.unique())\n\ndef test_layer (ID_N):\n    test_df2   = pd.DataFrame(Week, columns = [\"Weeks\"])\n    test_df_Y  = pd.DataFrame(Week, columns = [\"Weeks\"])\n    test_df_Y.insert(1, 'FVC', np.nan)\n    test_df2.insert(1, 'Percent', np.nan)\n    test_df2.insert(2, 'Age', np.nan)\n    test_df2.insert(3, 'Sex_Male', 0)\n    test_df2.insert(4, 'Sex_Female', 0)\n    test_df2.insert(5, 'Currently smokes', 0)\n    test_df2.insert(6, 'Ex-smoker', 0)\n    test_df2.insert(7, 'Never smoked', 0)\n\n    test_id = test_df.loc[test_df.Patient == Patient_list_test[ID_N]]\n    test_id = test_id.reset_index()\n\n    for i, D in enumerate(test_id.Weeks):\n        D = D + 12\n        if D <= 133:\n            test_df_Y.at[D, \"FVC\"] = test_id.FVC[i]\n            test_df2.at[D, \"Percent\"] = test_id.Percent[i]\n\n    test_df2.loc[:, \"Age\"] = test_id.Age[0]\n\n    if test_id.Sex[0] == \"Male\":\n        test_df2.loc[:, \"Sex_Male\"] = 1\n    else:\n        test_df2.loc[:, \"Sex_Female\"] = 1\n    \n    if test_id.SmokingStatus[0] == \"Currently smokes\":\n        test_df2.loc[:, \"Currently smokes\"] = 1\n    elif test_id.SmokingStatus[0] == \"Ex-smoker\":\n        test_df2.loc[:, \"Ex-smoker\"] = 1\n    else:\n        test_df2.loc[:, \"Never smoked\"] = 1\n        \n    test_df2 = test_df2.interpolate('linear', order=2, limit_direction='both')\n    test_df_Y = test_df_Y.interpolate('linear', order=2, limit_direction='both')\n    test_df_Y = test_df_Y.astype('int')\n    test_df_Y = test_df_Y.drop([\"Weeks\"], axis=1)\n    \n    return test_df2, test_df_Y","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.246736Z","iopub.execute_input":"2021-06-02T02:27:36.247042Z","iopub.status.idle":"2021-06-02T02:27:36.266906Z","shell.execute_reply.started":"2021-06-02T02:27:36.247014Z","shell.execute_reply":"2021-06-02T02:27:36.265651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_layer(0)[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.268406Z","iopub.execute_input":"2021-06-02T02:27:36.268810Z","iopub.status.idle":"2021-06-02T02:27:36.311454Z","shell.execute_reply.started":"2021-06-02T02:27:36.268777Z","shell.execute_reply":"2021-06-02T02:27:36.310313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_layer(0)[0].to_numpy()\nY_test = test_layer(0)[1].to_numpy() #Do not use Y_test\nsums= 0\n\nfor i in range(1, len(Patient_list_test)):\n    a = test_layer(i)[0].to_numpy()\n    X_test = np.append(X_test, a, axis=0)\n    \n    b = test_layer(i)[1].to_numpy()\n    Y_test = np.append(Y_test, b)    ","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.312798Z","iopub.execute_input":"2021-06-02T02:27:36.313113Z","iopub.status.idle":"2021-06-02T02:27:36.442512Z","shell.execute_reply.started":"2021-06-02T02:27:36.313081Z","shell.execute_reply":"2021-06-02T02:27:36.441562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Y_testは使わないが、一応作っておく","metadata":{}},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.443865Z","iopub.execute_input":"2021-06-02T02:27:36.444178Z","iopub.status.idle":"2021-06-02T02:27:36.449589Z","shell.execute_reply.started":"2021-06-02T02:27:36.444146Z","shell.execute_reply":"2021-06-02T02:27:36.448806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.450734Z","iopub.execute_input":"2021-06-02T02:27:36.451038Z","iopub.status.idle":"2021-06-02T02:27:36.463692Z","shell.execute_reply.started":"2021-06-02T02:27:36.451007Z","shell.execute_reply":"2021-06-02T02:27:36.462611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple Model","metadata":{}},{"cell_type":"code","source":"import sklearn\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn import tree \nimport lightgbm as lgb\n\ndef FitModel(X, Y, max_depth):\n    model = LogisticRegression(max_iter=max_depth, verbose=0)\n    model.fit(X, Y)\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.465064Z","iopub.execute_input":"2021-06-02T02:27:36.465398Z","iopub.status.idle":"2021-06-02T02:27:36.988845Z","shell.execute_reply.started":"2021-06-02T02:27:36.465368Z","shell.execute_reply":"2021-06-02T02:27:36.987775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainY = []\ntestY = []\n\n\"\"\"\nparams = {\n    'metric' : 'rmse',\n    'num_leaves': 100}\nlgb_train = lgb.Dataset(X_train, Y_train)\nmodel = lgb.train(params, lgb_train,)\n\"\"\"\nmodel = FitModel(X_train, Y_train, 100)\ntrainY.append(model.predict(X_train))\ntestY.append(model.predict(X_test))","metadata":{"execution":{"iopub.status.busy":"2021-06-02T02:27:36.990440Z","iopub.execute_input":"2021-06-02T02:27:36.990779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,6))\n\nY_train_Graph = pd.DataFrame(trainY[-1])\n#Y_train = Y_train.reset_index(drop=True)\nplt.plot(Y_train)\nplt.plot(Y_train_Graph, label = \"Predict\")\nplt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#DON'T USE THIS\n\"\"\"\nplt.figure(figsize=(18,8))\n\nID_NUM = 0\n\nfor i in range(5):\n    Y_test_Graph = pd.DataFrame(testY[i])\n    plt.plot(Y_test, linestyle = \"dashed\")\n    plt.plot(Y_test_Graph, label = \"Predict:{0}\".format(i))\n    plt.xlim(145*ID_NUM, 145*(ID_NUM+1))\n    \nplt.legend()\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#DON'T USE THIS\n'''\nimport math\nfrom scipy import stats\n\ndf_test = pd.DataFrame(testY)\nConfidence = []\n\nfor i in range(df_test.shape[1]):\n    data = np.array(df_test.iloc[:, i])\n    Confidence1 =  df_test.iloc[:, i].mean() - (2.086*np.std(data)/math.sqrt(21))\n    Confidence2 =  df_test.iloc[:, i].mean() + (2.086*np.std(data)/math.sqrt(21))\n    Confidence.append(Confidence2 - Confidence1)\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(Confidence)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission.csv","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame(columns = [\"Patient_Week\", \"FVC\", \"Confidence\"])\n\nWEEK = -12\nD = 0\n\nNUM = len(Patient_list_test)\n\nfor j in range(146):\n    \n    for i in range(NUM):\n        submission.loc[j*NUM + i,\"Patient_Week\"] = Patient_list_test[i] +\"_\" + str(WEEK)\n        submission.loc[j*NUM + i,\"FVC\"] = Y_test[i*146 + j]\n        submission.loc[j*NUM + i,\"Confidence\"] = 285\n\n    WEEK = WEEK + 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(40)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}