{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-01-01T14:13:31.481549Z","iopub.execute_input":"2022-01-01T14:13:31.481874Z","iopub.status.idle":"2022-01-01T14:13:31.50408Z","shell.execute_reply.started":"2022-01-01T14:13:31.48179Z","shell.execute_reply":"2022-01-01T14:13:31.503336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# images_path = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images'\ntrain_labels_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:15:46.319016Z","iopub.execute_input":"2022-01-01T14:15:46.319333Z","iopub.status.idle":"2022-01-01T14:15:46.375491Z","shell.execute_reply.started":"2022-01-01T14:15:46.319298Z","shell.execute_reply":"2022-01-01T14:15:46.374681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df[4:]","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:15:48.292688Z","iopub.execute_input":"2022-01-01T14:15:48.292988Z","iopub.status.idle":"2022-01-01T14:15:48.31529Z","shell.execute_reply.started":"2022-01-01T14:15:48.292954Z","shell.execute_reply":"2022-01-01T14:15:48.314312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df = train_labels_df[train_labels_df['Target'] == 1]\ntrain_labels_df","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:15:50.842749Z","iopub.execute_input":"2022-01-01T14:15:50.843027Z","iopub.status.idle":"2022-01-01T14:15:50.866587Z","shell.execute_reply.started":"2022-01-01T14:15:50.842997Z","shell.execute_reply":"2022-01-01T14:15:50.865851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df['x2'] = train_labels_df['x'] + train_labels_df['width']\ntrain_labels_df['y2'] = train_labels_df['y'] + train_labels_df['height']\ntrain_labels_df['T'] = np.where(train_labels_df['Target'] == 1, 'Pneumonia', '')\ntrain_labels_df","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:15:58.924974Z","iopub.execute_input":"2022-01-01T14:15:58.9255Z","iopub.status.idle":"2022-01-01T14:15:58.954593Z","shell.execute_reply.started":"2022-01-01T14:15:58.925462Z","shell.execute_reply":"2022-01-01T14:15:58.953812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df = train_labels_df.rename(columns={'x': 'x1', 'y': 'y1'})\ndel train_labels_df['width']\ndel train_labels_df['height']\ndel train_labels_df['Target']\ntrain_labels_df","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:16:01.834402Z","iopub.execute_input":"2022-01-01T14:16:01.834885Z","iopub.status.idle":"2022-01-01T14:16:01.8596Z","shell.execute_reply.started":"2022-01-01T14:16:01.83485Z","shell.execute_reply":"2022-01-01T14:16:01.858848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fun(s):\n    return f'train/{s}.jpg'\n#     return s.rsplit('.')[0]\n#     return s.split('/')[1]","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:16:04.474575Z","iopub.execute_input":"2022-01-01T14:16:04.474823Z","iopub.status.idle":"2022-01-01T14:16:04.479305Z","shell.execute_reply.started":"2022-01-01T14:16:04.474795Z","shell.execute_reply":"2022-01-01T14:16:04.478608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df['patientId'] = train_labels_df.apply(lambda row : fun(row['patientId']), axis = 1)\ntrain_labels_df","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:16:06.30273Z","iopub.execute_input":"2022-01-01T14:16:06.303291Z","iopub.status.idle":"2022-01-01T14:16:06.440097Z","shell.execute_reply.started":"2022-01-01T14:16:06.303254Z","shell.execute_reply":"2022-01-01T14:16:06.43935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df['x1'] = train_labels_df['x1'].astype(\"Int64\")\ntrain_labels_df['x2'] = train_labels_df['x2'].astype(\"Int64\")\ntrain_labels_df['y1'] = train_labels_df['y1'].astype(\"Int64\")\ntrain_labels_df['y2'] = train_labels_df['y2'].astype(\"Int64\")\ntrain_labels_df","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:16:09.366726Z","iopub.execute_input":"2022-01-01T14:16:09.367142Z","iopub.status.idle":"2022-01-01T14:16:09.396147Z","shell.execute_reply.started":"2022-01-01T14:16:09.367101Z","shell.execute_reply":"2022-01-01T14:16:09.395208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df","metadata":{"execution":{"iopub.status.busy":"2021-12-27T06:28:48.515495Z","iopub.execute_input":"2021-12-27T06:28:48.515754Z","iopub.status.idle":"2021-12-27T06:28:48.536139Z","shell.execute_reply.started":"2021-12-27T06:28:48.515723Z","shell.execute_reply":"2021-12-27T06:28:48.53531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn\nfrom sklearn.model_selection import train_test_split\n\ndf_train, df_valid = train_test_split(\n    train_labels_df,\n    test_size=0.1,\n    random_state=123, # Random seed\n    # stratify=train_labels_df['T'],\n    # shuffle=True, -> default: True\n) # 8590 Pneumonia\n","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:17:46.141455Z","iopub.execute_input":"2022-01-01T14:17:46.141702Z","iopub.status.idle":"2022-01-01T14:17:46.965668Z","shell.execute_reply.started":"2022-01-01T14:17:46.141676Z","shell.execute_reply":"2022-01-01T14:17:46.964964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:26:15.766499Z","iopub.execute_input":"2022-01-01T14:26:15.767257Z","iopub.status.idle":"2022-01-01T14:26:15.78791Z","shell.execute_reply.started":"2022-01-01T14:26:15.76721Z","shell.execute_reply":"2022-01-01T14:26:15.787204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 有: 6012 | 沒有: 20672 / unique(26684)\n- 有: 9555 | 沒有: 20672 / all(30332)\n- 1 epoch: 13341 iter ???(half of uniques)","metadata":{}},{"cell_type":"code","source":"df_train.to_csv('annot_train.csv', header=False, index=False)\ndf_valid.to_csv('annot_valid.csv', header=False, index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:20:05.687153Z","iopub.execute_input":"2022-01-01T14:20:05.68761Z","iopub.status.idle":"2022-01-01T14:20:05.741158Z","shell.execute_reply.started":"2022-01-01T14:20:05.687573Z","shell.execute_reply":"2022-01-01T14:20:05.74046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid","metadata":{"execution":{"iopub.status.busy":"2022-01-01T14:26:40.3501Z","iopub.execute_input":"2022-01-01T14:26:40.350797Z","iopub.status.idle":"2022-01-01T14:26:40.369173Z","shell.execute_reply.started":"2022-01-01T14:26:40.350753Z","shell.execute_reply":"2022-01-01T14:26:40.368474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Finish output train csv\n\n---\n\n### Start output inference csv","metadata":{}},{"cell_type":"code","source":"test_list = os.listdir('../input/rsna-pneumonia-detection-challenge/stage_2_test_images')\nfor i in range(len(test_list)):\n    pure_name = test_list[i].split('.')[0]\n    test_list[i] = f'dataset/val2017/{pure_name}.jpg'\n\ntest_list[:10]","metadata":{"execution":{"iopub.status.busy":"2021-12-26T08:09:09.796649Z","iopub.execute_input":"2021-12-26T08:09:09.797194Z","iopub.status.idle":"2021-12-26T08:09:10.134341Z","shell.execute_reply.started":"2021-12-26T08:09:09.797155Z","shell.execute_reply":"2021-12-26T08:09:10.133524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels_df = pd.DataFrame(data={\n    'patientId': test_list\n})\ntest_labels_df","metadata":{"execution":{"iopub.status.busy":"2021-12-26T08:09:46.649467Z","iopub.execute_input":"2021-12-26T08:09:46.649724Z","iopub.status.idle":"2021-12-26T08:09:46.663026Z","shell.execute_reply.started":"2021-12-26T08:09:46.649698Z","shell.execute_reply":"2021-12-26T08:09:46.662105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels_df['x1'] = np.NaN\ntest_labels_df['x2'] = np.NaN\ntest_labels_df['y1'] = np.NaN\ntest_labels_df['y2'] = np.NaN\ntest_labels_df['T'] = ''\ntest_labels_df","metadata":{"execution":{"iopub.status.busy":"2021-12-26T08:09:53.867553Z","iopub.execute_input":"2021-12-26T08:09:53.867819Z","iopub.status.idle":"2021-12-26T08:09:53.889653Z","shell.execute_reply.started":"2021-12-26T08:09:53.867771Z","shell.execute_reply":"2021-12-26T08:09:53.88869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels_df['x1'] = test_labels_df['x1'].astype(\"Int64\")\ntest_labels_df['x2'] = test_labels_df['x2'].astype(\"Int64\")\ntest_labels_df['y1'] = test_labels_df['y1'].astype(\"Int64\")\ntest_labels_df['y2'] = test_labels_df['y2'].astype(\"Int64\")\ntest_labels_df","metadata":{"execution":{"iopub.status.busy":"2021-12-26T08:09:56.127194Z","iopub.execute_input":"2021-12-26T08:09:56.127444Z","iopub.status.idle":"2021-12-26T08:09:56.151056Z","shell.execute_reply.started":"2021-12-26T08:09:56.127418Z","shell.execute_reply":"2021-12-26T08:09:56.150454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels_df.to_csv('annot_test.csv', header=False, index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-26T08:10:01.636565Z","iopub.execute_input":"2021-12-26T08:10:01.636927Z","iopub.status.idle":"2021-12-26T08:10:01.656305Z","shell.execute_reply.started":"2021-12-26T08:10:01.63689Z","shell.execute_reply":"2021-12-26T08:10:01.655684Z"},"trusted":true},"execution_count":null,"outputs":[]}]}