{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":14220928,"sourceType":"datasetVersion","datasetId":9071576}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/train-and-test-data'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:50.052731Z","iopub.execute_input":"2025-12-19T14:34:50.053226Z","iopub.status.idle":"2025-12-19T14:34:50.060339Z","shell.execute_reply.started":"2025-12-19T14:34:50.053191Z","shell.execute_reply":"2025-12-19T14:34:50.059316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:50.062452Z","iopub.execute_input":"2025-12-19T14:34:50.062815Z","iopub.status.idle":"2025-12-19T14:34:50.080696Z","shell.execute_reply.started":"2025-12-19T14:34:50.062785Z","shell.execute_reply":"2025-12-19T14:34:50.079590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/train-and-test-data/train_clean.csv')\ntest_df = pd.read_csv('/kaggle/input/train-and-test-data/test_clean.csv')\n\ntrain_df.shape\ntest_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:50.081962Z","iopub.execute_input":"2025-12-19T14:34:50.082249Z","iopub.status.idle":"2025-12-19T14:34:50.118416Z","shell.execute_reply.started":"2025-12-19T14:34:50.082224Z","shell.execute_reply":"2025-12-19T14:34:50.117539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.drop(['id', 'sii'], axis=1)\ny = train_df['sii']\n\nX_test = test_df.drop(['id'], axis=1)\n\ntest_ids = test_df['id'].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:50.119648Z","iopub.execute_input":"2025-12-19T14:34:50.119986Z","iopub.status.idle":"2025-12-19T14:34:50.126763Z","shell.execute_reply.started":"2025-12-19T14:34:50.119957Z","shell.execute_reply":"2025-12-19T14:34:50.125932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf = RandomForestClassifier(\n    n_estimators=300,      # 樹的數量\n    random_state=42,\n    class_weight= 'balanced'\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:50.128621Z","iopub.execute_input":"2025-12-19T14:34:50.128934Z","iopub.status.idle":"2025-12-19T14:34:50.145252Z","shell.execute_reply.started":"2025-12-19T14:34:50.128905Z","shell.execute_reply":"2025-12-19T14:34:50.144314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf.fit(X, y)\ny_test_pred = rf.predict(X_test)\n\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'sii': y_test_pred.astype(int)\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:50.146303Z","iopub.execute_input":"2025-12-19T14:34:50.146688Z","iopub.status.idle":"2025-12-19T14:34:51.551273Z","shell.execute_reply.started":"2025-12-19T14:34:50.146657Z","shell.execute_reply":"2025-12-19T14:34:51.550565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:51.552262Z","iopub.execute_input":"2025-12-19T14:34:51.552528Z","iopub.status.idle":"2025-12-19T14:34:51.560977Z","shell.execute_reply.started":"2025-12-19T14:34:51.552502Z","shell.execute_reply":"2025-12-19T14:34:51.560128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(os.path.join('/kaggle/working', 'submission.csv'), index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T14:34:51.562089Z","iopub.execute_input":"2025-12-19T14:34:51.562423Z","iopub.status.idle":"2025-12-19T14:34:51.579838Z","shell.execute_reply.started":"2025-12-19T14:34:51.562388Z","shell.execute_reply":"2025-12-19T14:34:51.579091Z"}},"outputs":[],"execution_count":null}]}