{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:30.066109Z","iopub.execute_input":"2024-12-05T09:48:30.066744Z","iopub.status.idle":"2024-12-05T09:48:31.364826Z","shell.execute_reply.started":"2024-12-05T09:48:30.066705Z","shell.execute_reply":"2024-12-05T09:48:31.363686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:31.366881Z","iopub.execute_input":"2024-12-05T09:48:31.367936Z","iopub.status.idle":"2024-12-05T09:48:46.371717Z","shell.execute_reply.started":"2024-12-05T09:48:31.367865Z","shell.execute_reply":"2024-12-05T09:48:46.370591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Dataset\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.373678Z","iopub.execute_input":"2024-12-05T09:48:46.374472Z","iopub.status.idle":"2024-12-05T09:48:46.440795Z","shell.execute_reply.started":"2024-12-05T09:48:46.374425Z","shell.execute_reply":"2024-12-05T09:48:46.439640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Selection (Pilih fitur yang relevan untuk model)\nfeatures = ['Physical-BMI', 'Physical-Height', 'Physical-Weight']  # Sesuaikan dengan dataset Anda\ntarget = 'sii'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.443782Z","iopub.execute_input":"2024-12-05T09:48:46.444283Z","iopub.status.idle":"2024-12-05T09:48:46.449960Z","shell.execute_reply.started":"2024-12-05T09:48:46.444234Z","shell.execute_reply":"2024-12-05T09:48:46.448565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle Missing Values with Imputation\nfrom sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.451706Z","iopub.execute_input":"2024-12-05T09:48:46.452088Z","iopub.status.idle":"2024-12-05T09:48:46.466182Z","shell.execute_reply.started":"2024-12-05T09:48:46.452056Z","shell.execute_reply":"2024-12-05T09:48:46.464455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = SimpleImputer(strategy='mean')  # Untuk fitur numerik\ntrain[features] = imputer.fit_transform(train[features])\ntest[features] = imputer.transform(test[features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.468372Z","iopub.execute_input":"2024-12-05T09:48:46.469074Z","iopub.status.idle":"2024-12-05T09:48:46.492704Z","shell.execute_reply.started":"2024-12-05T09:48:46.469010Z","shell.execute_reply":"2024-12-05T09:48:46.491428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tangani NaN dan inf pada kolom target 'sii' di train\ntrain[target] = train[target].fillna(0)  # Ganti NaN dengan 0 atau nilai yang sesuai\ntrain[target] = train[target].replace([np.inf, -np.inf], 0)  # Ganti inf dengan 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.494336Z","iopub.execute_input":"2024-12-05T09:48:46.495320Z","iopub.status.idle":"2024-12-05T09:48:46.509283Z","shell.execute_reply.started":"2024-12-05T09:48:46.495267Z","shell.execute_reply":"2024-12-05T09:48:46.508155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pastikan target menjadi integer\ny_train = train[target].astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.510717Z","iopub.execute_input":"2024-12-05T09:48:46.511159Z","iopub.status.idle":"2024-12-05T09:48:46.526237Z","shell.execute_reply.started":"2024-12-05T09:48:46.511115Z","shell.execute_reply":"2024-12-05T09:48:46.524926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalize Data\nscaler = StandardScaler()\nX = scaler.fit_transform(train[features])\ny = y_train  # Menggunakan target yang telah dibersihkan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.527816Z","iopub.execute_input":"2024-12-05T09:48:46.528284Z","iopub.status.idle":"2024-12-05T09:48:46.543201Z","shell.execute_reply.started":"2024-12-05T09:48:46.528237Z","shell.execute_reply":"2024-12-05T09:48:46.541857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split Data\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.547081Z","iopub.execute_input":"2024-12-05T09:48:46.547524Z","iopub.status.idle":"2024-12-05T09:48:46.555329Z","shell.execute_reply.started":"2024-12-05T09:48:46.547480Z","shell.execute_reply":"2024-12-05T09:48:46.554126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the Keras Model\nmodel = Sequential()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.556769Z","iopub.execute_input":"2024-12-05T09:48:46.557188Z","iopub.status.idle":"2024-12-05T09:48:46.568494Z","shell.execute_reply.started":"2024-12-05T09:48:46.557145Z","shell.execute_reply":"2024-12-05T09:48:46.567319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Input Layer\nmodel.add(Dense(128, input_dim=X_train.shape[1], activation='relu'))  # 128 neurons in the input layer\nmodel.add(BatchNormalization())  # Batch Normalization for better convergence","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.569854Z","iopub.execute_input":"2024-12-05T09:48:46.570302Z","iopub.status.idle":"2024-12-05T09:48:46.667992Z","shell.execute_reply.started":"2024-12-05T09:48:46.570251Z","shell.execute_reply":"2024-12-05T09:48:46.666795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hidden Layers\nmodel.add(Dense(64, activation='relu'))  # 64 neurons in the hidden layer\nmodel.add(Dropout(0.5))  # Dropout to prevent overfitting\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.669315Z","iopub.execute_input":"2024-12-05T09:48:46.669641Z","iopub.status.idle":"2024-12-05T09:48:46.695484Z","shell.execute_reply.started":"2024-12-05T09:48:46.669602Z","shell.execute_reply":"2024-12-05T09:48:46.694382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.add(Dense(32, activation='relu'))  # 32 neurons in another hidden layer\nmodel.add(Dropout(0.5))  # Dropout again","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.696725Z","iopub.execute_input":"2024-12-05T09:48:46.697058Z","iopub.status.idle":"2024-12-05T09:48:46.729454Z","shell.execute_reply.started":"2024-12-05T09:48:46.697027Z","shell.execute_reply":"2024-12-05T09:48:46.728381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Output Layer (assuming a classification task with integers)\nmodel.add(Dense(1, activation='sigmoid'))  # Output layer with sigmoid activation for binary classification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.733560Z","iopub.execute_input":"2024-12-05T09:48:46.734017Z","iopub.status.idle":"2024-12-05T09:48:46.757758Z","shell.execute_reply.started":"2024-12-05T09:48:46.733973Z","shell.execute_reply":"2024-12-05T09:48:46.756770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer=Adam(), loss='binary_crossentropy', metrics=['accuracy'])\n\n# EarlyStopping to avoid overfitting\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Train the model\nhistory = model.fit(X_train, y_train, epochs=100, batch_size=32, validation_data=(X_val, y_val), callbacks=[early_stopping], verbose=1)\n\n# Evaluate Model using Cohen Kappa Score\ny_pred = model.predict(X_val)\ny_pred_class = (y_pred > 0.5).astype(int)  # Convert probabilities to binary labels (0 or 1)\n\nscore = cohen_kappa_score(y_val, y_pred_class)\nprint(\"Quadratic Weighted Kappa Score (Keras):\", score)\n\n# Make Predictions on Test Data using the trained model\nX_test = scaler.transform(test[features])\ntest['sii'] = model.predict(X_test)\ntest['sii'] = (test['sii'] > 0.5).astype(int)  # Convert probabilities to binary labels (0 or 1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:46.759536Z","iopub.execute_input":"2024-12-05T09:48:46.759963Z","iopub.status.idle":"2024-12-05T09:48:53.934852Z","shell.execute_reply.started":"2024-12-05T09:48:46.759919Z","shell.execute_reply":"2024-12-05T09:48:53.933950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save Submission File\nsubmission = test[['id', 'sii']]  # Hanya kolom 'id' dan 'sii'\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file saved as 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:48:53.936187Z","iopub.execute_input":"2024-12-05T09:48:53.936525Z","iopub.status.idle":"2024-12-05T09:48:53.944988Z","shell.execute_reply.started":"2024-12-05T09:48:53.936492Z","shell.execute_reply":"2024-12-05T09:48:53.943813Z"}},"outputs":[],"execution_count":null}]}