{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:11:20.567666Z","iopub.execute_input":"2024-12-05T13:11:20.570435Z","iopub.status.idle":"2024-12-05T13:11:25.207846Z","shell.execute_reply.started":"2024-12-05T13:11:20.570355Z","shell.execute_reply":"2024-12-05T13:11:25.206713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder, LabelBinarizer\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.metrics import classification_report, confusion_matrix, cohen_kappa_score\n\n# Fungsi untuk menghitung Quadratic Weighted Kappa\ndef quadratic_weighted_kappa(y_true, y_pred):\n    # Hitung histogram prediksi dan aktual\n    o = confusion_matrix(y_true, y_pred)\n    n = len(np.unique(y_true))\n    \n    # Matriks bobot W\n    weight_matrix = np.zeros((n, n))\n    for i in range(n):\n        for j in range(n):\n            weight_matrix[i, j] = (i - j) ** 2 / (n - 1) ** 2\n    \n    # Matriks E (ekspektasi)\n    o_hist = np.histogram(y_true, bins=np.arange(n+1))[0]\n    e_hist = np.histogram(y_pred, bins=np.arange(n+1))[0]\n    e = np.outer(o_hist, e_hist) / np.sum(o_hist) / np.sum(e_hist)\n    \n    # Hitung QWK\n    numerator = np.sum(weight_matrix * o)\n    denominator = np.sum(weight_matrix * e)\n    kappa = 1 - (numerator / denominator)\n    \n    return kappa\n\n# Load dataset\ntrain_csv_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\nif os.path.exists(train_csv_path):\n    train_data = pd.read_csv(train_csv_path)\nelse:\n    raise FileNotFoundError(\"Dataset train.csv tidak ditemukan.\")\n\n# Tampilkan informasi dataset\nprint(f\"Dimensi dataset train.csv: {train_data.shape}\")\nprint(f\"Jumlah missing values pada dataset:\\n{train_data.isnull().sum()}\")\n\n# Tangani missing values\nfor col in train_data.columns:\n    if train_data[col].isnull().sum() > 0:\n        if train_data[col].dtype in ['float64', 'int64']:\n            train_data[col] = train_data[col].fillna(train_data[col].median())  # Isi missing dengan median\n        else:\n            train_data[col] = train_data[col].fillna('unknown')  # Isi missing dengan kategori 'unknown'\n\n# Pisahkan fitur dan target\nX = train_data.drop(columns=['id', 'sii'])  # Hilangkan kolom id dan target\ny = train_data['sii']\n\n# Encoding untuk data kategorikal\ncategorical_cols = X.select_dtypes(include=['object']).columns  # Pilih kolom kategorikal\nlabel_encoders = {}\nfor col in categorical_cols:\n    label_encoders[col] = LabelEncoder()\n    X[col] = label_encoders[col].fit_transform(X[col])  # Transform kolom menjadi numerik\n\n# Simpan kolom fitur untuk memastikan konsistensi\nfeature_columns = X.columns.tolist()\n\n# Split data untuk pelatihan dan validasi\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\n# Feature scaling untuk data numerik\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_val = scaler.transform(X_val)\n\n# Definisikan model Perceptron\nclf = Perceptron(random_state=42)\n\n# Latih model\nclf.fit(X_train, y_train)\n\n# Evaluasi model\ny_pred = clf.predict(X_val)\nprint(\"Classification Report pada data validasi:\")\nprint(classification_report(y_val, y_pred))\nprint(\"Confusion Matrix pada data validasi:\")\nprint(confusion_matrix(y_val, y_pred))\n\n# Hitung Quadratic Weighted Kappa\nqwk = quadratic_weighted_kappa(y_val, y_pred)\nprint(f\"Quadratic Weighted Kappa (QWK): {qwk:.4f}\")\n\n# Prediksi pada data uji\ntest_csv_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\nif os.path.exists(test_csv_path):\n    test_data = pd.read_csv(test_csv_path)\n    test_ids = test_data['id']  # Simpan ID untuk file submission\n    test_data = test_data.drop(columns=['id'])\n    \n    # Tangani missing values pada data uji\n    for col in test_data.columns:\n        if col in categorical_cols:\n            if col in label_encoders:\n                test_data[col] = test_data[col].fillna('unknown')\n                test_data[col] = label_encoders[col].transform(test_data[col])  # Transform kategori ke numerik\n        elif test_data[col].dtype in ['float64', 'int64']:\n            test_data[col] = test_data[col].fillna(test_data[col].median())\n    \n    # Pastikan urutan kolom fitur sesuai dengan data pelatihan\n    test_data = test_data.reindex(columns=feature_columns, fill_value=0)  # Kolom yang hilang diisi dengan nilai 0\n\n    # Scaling data uji\n    test_data = scaler.transform(test_data)\n    \n    # Prediksi\n    test_predictions = clf.predict(test_data)\n    \n    # Buat file submission\n    submission = pd.DataFrame({'id': test_ids, 'sii': test_predictions})\n    submission.to_csv('submission.csv', index=False)\n    print(\"File submission.csv telah dibuat.\")\nelse:\n    print(\"Dataset test.csv tidak ditemukan.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:48:19.887256Z","iopub.execute_input":"2024-12-05T13:48:19.887630Z","iopub.status.idle":"2024-12-05T13:48:20.142298Z","shell.execute_reply.started":"2024-12-05T13:48:19.887599Z","shell.execute_reply":"2024-12-05T13:48:20.141096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}