{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# 檢查 /kaggle/input 資料夾，確認文件是否存在\nfile_found = False\nfile_name = \"data_dictionary.csv\"  # 更新為您的文件名\nfolder_name = \"data-dict\"  # 更新為您的資料夾名稱\n\n# 檢查目錄\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename == file_name:\n            file_found = True\n            file_path = os.path.join(dirname, filename)\n            print(f\"Reading file: {file_path}\")\n            \n            # 處理文件\n            try:\n                # 加載數據\n                data = pd.read_csv(file_path)\n                print(\"File loaded successfully:\")\n                print(data.head())  # 打印數據的前幾行\n\n                # 示例處理：提取第一列並添加新列\n                if data.shape[1] > 0:\n                    processed_data = data.iloc[:, [0]].copy()  # 提取第一列\n                    processed_data['sii'] = 0  # 添加新列 'sii'\n                    print(\"Processed data:\")\n                    print(processed_data.head())\n                else:\n                    print(\"Error: The file does not contain enough columns.\")\n                    processed_data = None\n\n                # 保存為 submission.csv\n                if processed_data is not None:\n                    output_path = '/kaggle/working/submission.csv'\n                    processed_data.to_csv(output_path, index=False)\n                    print(f\"Submission file saved to: {output_path}\")\n            \n            except Exception as e:\n                print(f\"Error processing the file: {e}\")\n            break\n\nif not file_found:\n    print(f\"Error: '{file_name}' not found in the '{folder_name}' directory.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T13:57:32.460609Z","iopub.execute_input":"2024-12-04T13:57:32.460998Z","iopub.status.idle":"2024-12-04T13:57:37.473802Z","shell.execute_reply.started":"2024-12-04T13:57:32.460960Z","shell.execute_reply":"2024-12-04T13:57:37.472342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nfile_name = \"data_dictionary.csv\"\nfolder_name = \"data-dict\"\nfile_found = False\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename == file_name:\n            file_found = True\n            file_path = os.path.join(dirname, filename)\n            print(f\"Reading file: {file_path}\")\n            \n            try:\n                # 加載數據\n                data = pd.read_csv(file_path)\n                print(\"File loaded successfully:\")\n                print(data.head())\n\n                # 處理數據\n                if data.shape[1] > 0:\n                    processed_data = data.iloc[:, [0]].copy()\n                    processed_data.rename(columns={processed_data.columns[0]: 'id'}, inplace=True)\n                    processed_data['prediction'] = 0  # 添加新列\n                else:\n                    raise ValueError(\"File does not contain enough columns.\")\n\n                # 填充空值並確保類型正確\n                processed_data.fillna(0, inplace=True)\n                processed_data['id'] = processed_data['id'].astype(int)\n                processed_data['prediction'] = processed_data['prediction'].astype(float)\n\n                # 保存文件\n                output_path = '/kaggle/working/submission.csv'\n                processed_data.to_csv(output_path, index=False)\n                print(f\"Submission file saved to: {output_path}\")\n\n            except Exception as e:\n                print(f\"Error processing the file: {e}\")\n            break\n\nif not file_found:\n    print(f\"Error: '{file_name}' not found in the '{folder_name}' directory.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:12:55.114899Z","iopub.execute_input":"2024-12-04T14:12:55.115346Z","iopub.status.idle":"2024-12-04T14:12:56.360534Z","shell.execute_reply.started":"2024-12-04T14:12:55.115306Z","shell.execute_reply":"2024-12-04T14:12:56.359393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nfile_name = \"data_dictionary.csv\"\nfolder_name = \"child-mind-institute-problematic-internet-use\"\nfile_found = False\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename == file_name:\n            file_found = True\n            file_path = os.path.join(dirname, filename)\n            print(f\"Reading file: {file_path}\")\n            \n            try:\n                # 加載數據\n                data = pd.read_csv(file_path)\n                print(\"File loaded successfully:\")\n                print(data.head())\n\n                # 處理數據\n                if data.shape[1] > 0:\n                    processed_data = data.iloc[:, [0]].copy()  # 提取第一列\n                    processed_data.rename(columns={processed_data.columns[0]: 'id'}, inplace=True)\n                    \n                    # 過濾非數字行\n                    processed_data = processed_data[processed_data['id'].apply(lambda x: str(x).isdigit())]\n                    processed_data['id'] = processed_data['id'].astype(int)  # 確保 'id' 是整數\n                    processed_data['prediction'] = 0  # 添加新列\n                else:\n                    raise ValueError(\"File does not contain enough columns.\")\n\n                # 填充空值並保存\n                processed_data.fillna(0, inplace=True)\n                output_path = '/kaggle/working/submission.csv'\n                processed_data.to_csv(output_path, index=False)\n                print(f\"Submission file saved to: {output_path}\")\n\n            except Exception as e:\n                print(f\"Error processing the file: {e}\")\n            break\n\nif not file_found:\n    print(f\"Error: '{file_name}' not found in the '{folder_name}' directory.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T14:14:43.167192Z","iopub.execute_input":"2024-12-04T14:14:43.167604Z","iopub.status.idle":"2024-12-04T14:14:44.381932Z","shell.execute_reply.started":"2024-12-04T14:14:43.167569Z","shell.execute_reply":"2024-12-04T14:14:44.380757Z"}},"outputs":[],"execution_count":null}]}