{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:01.279251Z","iopub.execute_input":"2024-12-05T13:53:01.280882Z","iopub.status.idle":"2024-12-05T13:53:02.337247Z","shell.execute_reply.started":"2024-12-05T13:53:01.280830Z","shell.execute_reply":"2024-12-05T13:53:02.336174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import library yang diperlukan\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:02.339718Z","iopub.execute_input":"2024-12-05T13:53:02.340197Z","iopub.status.idle":"2024-12-05T13:53:03.329659Z","shell.execute_reply.started":"2024-12-05T13:53:02.340148Z","shell.execute_reply":"2024-12-05T13:53:03.328449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load dataset tabular\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# Melihat beberapa baris data\nprint(\"Train Data:\")\nprint(train_df.head())\n\nprint(\"\\nTest Data:\")\nprint(test_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.330952Z","iopub.execute_input":"2024-12-05T13:53:03.331457Z","iopub.status.idle":"2024-12-05T13:53:03.437285Z","shell.execute_reply.started":"2024-12-05T13:53:03.331421Z","shell.execute_reply":"2024-12-05T13:53:03.436049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_df.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.438423Z","iopub.execute_input":"2024-12-05T13:53:03.438798Z","iopub.status.idle":"2024-12-05T13:53:03.449621Z","shell.execute_reply.started":"2024-12-05T13:53:03.438761Z","shell.execute_reply":"2024-12-05T13:53:03.448459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df.drop(columns=['id', 'sii'])\ny = train_df['sii']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.452903Z","iopub.execute_input":"2024-12-05T13:53:03.453833Z","iopub.status.idle":"2024-12-05T13:53:03.460614Z","shell.execute_reply.started":"2024-12-05T13:53:03.453769Z","shell.execute_reply":"2024-12-05T13:53:03.459727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_cols_in_test = set(X.columns) - set(test_df.columns)\nX = X.drop(columns=missing_cols_in_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.461873Z","iopub.execute_input":"2024-12-05T13:53:03.462350Z","iopub.status.idle":"2024-12-05T13:53:03.471754Z","shell.execute_reply.started":"2024-12-05T13:53:03.462300Z","shell.execute_reply":"2024-12-05T13:53:03.470383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_cols = X.select_dtypes(include=['int64', 'float64']).columns\ncategorical_cols = X.select_dtypes(include=['object']).columns\n\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median'))\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.473019Z","iopub.execute_input":"2024-12-05T13:53:03.473395Z","iopub.status.idle":"2024-12-05T13:53:03.487011Z","shell.execute_reply.started":"2024-12-05T13:53:03.473360Z","shell.execute_reply":"2024-12-05T13:53:03.485885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_processed = preprocessor.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.488505Z","iopub.execute_input":"2024-12-05T13:53:03.489006Z","iopub.status.idle":"2024-12-05T13:53:03.545017Z","shell.execute_reply.started":"2024-12-05T13:53:03.488940Z","shell.execute_reply":"2024-12-05T13:53:03.543918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X_processed, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.546342Z","iopub.execute_input":"2024-12-05T13:53:03.546655Z","iopub.status.idle":"2024-12-05T13:53:03.555751Z","shell.execute_reply.started":"2024-12-05T13:53:03.546625Z","shell.execute_reply":"2024-12-05T13:53:03.554418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestRegressor(n_estimators=100, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.557141Z","iopub.execute_input":"2024-12-05T13:53:03.557624Z","iopub.status.idle":"2024-12-05T13:53:03.566175Z","shell.execute_reply.started":"2024-12-05T13:53:03.557573Z","shell.execute_reply":"2024-12-05T13:53:03.564835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:03.567520Z","iopub.execute_input":"2024-12-05T13:53:03.567867Z","iopub.status.idle":"2024-12-05T13:53:07.885685Z","shell.execute_reply.started":"2024-12-05T13:53:03.567824Z","shell.execute_reply":"2024-12-05T13:53:07.884398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_val)\nmse = mean_squared_error(y_val, y_pred)\nprint(f'Mean Squared Error: {mse}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:07.887399Z","iopub.execute_input":"2024-12-05T13:53:07.887727Z","iopub.status.idle":"2024-12-05T13:53:07.908335Z","shell.execute_reply.started":"2024-12-05T13:53:07.887698Z","shell.execute_reply":"2024-12-05T13:53:07.907029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_df.drop(columns=['id'])\nX_test = X_test.reindex(columns=X.columns, fill_value=np.nan)\nX_test_processed = preprocessor.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:07.909739Z","iopub.execute_input":"2024-12-05T13:53:07.910210Z","iopub.status.idle":"2024-12-05T13:53:07.925420Z","shell.execute_reply.started":"2024-12-05T13:53:07.910159Z","shell.execute_reply":"2024-12-05T13:53:07.924274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = model.predict(X_test_processed)\ntest_df['sii'] = test_predictions.round().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:07.929450Z","iopub.execute_input":"2024-12-05T13:53:07.929810Z","iopub.status.idle":"2024-12-05T13:53:07.952981Z","shell.execute_reply.started":"2024-12-05T13:53:07.929778Z","shell.execute_reply":"2024-12-05T13:53:07.951598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = test_df[['id', 'sii']]\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:53:07.954592Z","iopub.execute_input":"2024-12-05T13:53:07.955132Z","iopub.status.idle":"2024-12-05T13:53:07.969323Z","shell.execute_reply.started":"2024-12-05T13:53:07.955060Z","shell.execute_reply":"2024-12-05T13:53:07.967868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}