{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # Problematic Internet Uselinear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-03T19:49:44.042925Z","iopub.execute_input":"2024-10-03T19:49:44.043311Z","iopub.status.idle":"2024-10-03T19:49:47.900510Z","shell.execute_reply.started":"2024-10-03T19:49:44.043245Z","shell.execute_reply":"2024-10-03T19:49:47.899252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:40:45.597411Z","iopub.execute_input":"2024-10-03T20:40:45.598408Z","iopub.status.idle":"2024-10-03T20:40:45.604351Z","shell.execute_reply.started":"2024-10-03T20:40:45.598365Z","shell.execute_reply":"2024-10-03T20:40:45.603182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:41:13.679140Z","iopub.execute_input":"2024-10-03T20:41:13.679558Z","iopub.status.idle":"2024-10-03T20:41:13.737192Z","shell.execute_reply.started":"2024-10-03T20:41:13.679521Z","shell.execute_reply":"2024-10-03T20:41:13.736406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inspect the columns of train_data\nprint(\"Train Data Columns:\")\nprint(train_data.columns.tolist())\n\n# Inspect the first few rows\nprint(\"\\nTrain Data Head:\")\nprint(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:43:36.048445Z","iopub.execute_input":"2024-10-03T20:43:36.049403Z","iopub.status.idle":"2024-10-03T20:43:36.072941Z","shell.execute_reply.started":"2024-10-03T20:43:36.049349Z","shell.execute_reply":"2024-10-03T20:43:36.071904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_col = 'sii'  \n\nif target_col in train_data.columns:\n    feature_cols_train = train_data.drop(columns=[target_col]).columns.tolist()\n    y = train_data[target_col]\n    print(f\"\\nUsing '{target_col}' as the target variable.\")\nelse:\n    raise KeyError(f\"The target column '{target_col}' does not exist in train_data.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:44:21.328514Z","iopub.execute_input":"2024-10-03T20:44:21.329403Z","iopub.status.idle":"2024-10-03T20:44:21.338888Z","shell.execute_reply.started":"2024-10-03T20:44:21.329353Z","shell.execute_reply":"2024-10-03T20:44:21.337941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_targets = y.isnull().sum()\nprint(f\"Number of missing values in '{target_col}': {missing_targets}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:44:51.761865Z","iopub.execute_input":"2024-10-03T20:44:51.762348Z","iopub.status.idle":"2024-10-03T20:44:51.768523Z","shell.execute_reply.started":"2024-10-03T20:44:51.762302Z","shell.execute_reply":"2024-10-03T20:44:51.767162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if missing_targets > 0:\n    print(f\"Removing {missing_targets} rows with missing target values.\")\n    train_data = train_data.dropna(subset=[target_col])\n    y = train_data[target_col]\n    X = train_data.drop(columns=[target_col])\nelse:\n    X = train_data.drop(columns=[target_col])","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:45:21.777614Z","iopub.execute_input":"2024-10-03T20:45:21.778456Z","iopub.status.idle":"2024-10-03T20:45:21.791034Z","shell.execute_reply.started":"2024-10-03T20:45:21.778415Z","shell.execute_reply":"2024-10-03T20:45:21.789917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols_test = test_data.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:45:59.974756Z","iopub.execute_input":"2024-10-03T20:45:59.975206Z","iopub.status.idle":"2024-10-03T20:45:59.980287Z","shell.execute_reply.started":"2024-10-03T20:45:59.975167Z","shell.execute_reply":"2024-10-03T20:45:59.979075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_in_test = set(feature_cols_train) - set(feature_cols_test)\nprint(f\"\\nColumns missing in test_data: {missing_in_test}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:46:22.820558Z","iopub.execute_input":"2024-10-03T20:46:22.821414Z","iopub.status.idle":"2024-10-03T20:46:22.826795Z","shell.execute_reply.started":"2024-10-03T20:46:22.821370Z","shell.execute_reply":"2024-10-03T20:46:22.825670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in missing_in_test:\n    test_data[col] = np.nan","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:46:51.011994Z","iopub.execute_input":"2024-10-03T20:46:51.012456Z","iopub.status.idle":"2024-10-03T20:46:51.023664Z","shell.execute_reply.started":"2024-10-03T20:46:51.012413Z","shell.execute_reply":"2024-10-03T20:46:51.022593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data[feature_cols_train]","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:47:14.352672Z","iopub.execute_input":"2024-10-03T20:47:14.353361Z","iopub.status.idle":"2024-10-03T20:47:14.360136Z","shell.execute_reply.started":"2024-10-03T20:47:14.353320Z","shell.execute_reply":"2024-10-03T20:47:14.359181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_cols = X.select_dtypes(include=[np.number]).columns.tolist()\ncategorical_cols = X.select_dtypes(include=['object', 'category']).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:47:40.000791Z","iopub.execute_input":"2024-10-03T20:47:40.001489Z","iopub.status.idle":"2024-10-03T20:47:40.008955Z","shell.execute_reply.started":"2024-10-03T20:47:40.001450Z","shell.execute_reply":"2024-10-03T20:47:40.008001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\nNumeric Columns: {numeric_cols}\")\nprint(f\"Categorical Columns: {categorical_cols}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:48:03.422745Z","iopub.execute_input":"2024-10-03T20:48:03.423676Z","iopub.status.idle":"2024-10-03T20:48:03.428565Z","shell.execute_reply.started":"2024-10-03T20:48:03.423635Z","shell.execute_reply":"2024-10-03T20:48:03.427535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[numeric_cols] = X[numeric_cols].fillna(X[numeric_cols].median())\ntest_data[numeric_cols] = test_data[numeric_cols].fillna(X[numeric_cols].median())","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:48:41.767740Z","iopub.execute_input":"2024-10-03T20:48:41.768153Z","iopub.status.idle":"2024-10-03T20:48:41.863501Z","shell.execute_reply.started":"2024-10-03T20:48:41.768115Z","shell.execute_reply":"2024-10-03T20:48:41.862684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in categorical_cols:\n    mode = X[col].mode()[0] if not X[col].mode().empty else 'Unknown'\n    X[col] = X[col].fillna(mode)\n    test_data[col] = test_data[col].fillna(mode)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:49:04.964417Z","iopub.execute_input":"2024-10-03T20:49:04.965522Z","iopub.status.idle":"2024-10-03T20:49:05.005135Z","shell.execute_reply.started":"2024-10-03T20:49:04.965456Z","shell.execute_reply":"2024-10-03T20:49:05.004284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\n\nfor col in categorical_cols:\n    X[col] = le.fit_transform(X[col].astype(str))\n    classes = list(le.classes_)\n    test_data[col] = test_data[col].apply(lambda x: x if x in classes else '<unknown>')\n    if '<unknown>' not in classes:\n        le.classes_ = np.append(le.classes_, '<unknown>')\n    test_data[col] = le.transform(test_data[col].astype(str))","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:52:35.286801Z","iopub.execute_input":"2024-10-03T20:52:35.287234Z","iopub.status.idle":"2024-10-03T20:52:35.341201Z","shell.execute_reply.started":"2024-10-03T20:52:35.287200Z","shell.execute_reply":"2024-10-03T20:52:35.340047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X\ny = y\n\nassert list(X.columns) == list(test_data.columns), \"Mismatch in feature columns between train and test data.\"","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:53:16.959373Z","iopub.execute_input":"2024-10-03T20:53:16.960134Z","iopub.status.idle":"2024-10-03T20:53:16.964981Z","shell.execute_reply.started":"2024-10-03T20:53:16.960096Z","shell.execute_reply":"2024-10-03T20:53:16.963821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:54:15.655657Z","iopub.execute_input":"2024-10-03T20:54:15.656366Z","iopub.status.idle":"2024-10-03T20:54:15.670460Z","shell.execute_reply.started":"2024-10-03T20:54:15.656324Z","shell.execute_reply":"2024-10-03T20:54:15.669302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\nMissing values in y_train: {y_train.isnull().sum()}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:54:37.756965Z","iopub.execute_input":"2024-10-03T20:54:37.757411Z","iopub.status.idle":"2024-10-03T20:54:37.762878Z","shell.execute_reply.started":"2024-10-03T20:54:37.757368Z","shell.execute_reply":"2024-10-03T20:54:37.761914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:56:01.743376Z","iopub.execute_input":"2024-10-03T20:56:01.744069Z","iopub.status.idle":"2024-10-03T20:56:02.401418Z","shell.execute_reply.started":"2024-10-03T20:56:01.744030Z","shell.execute_reply":"2024-10-03T20:56:02.400281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:56:35.576838Z","iopub.execute_input":"2024-10-03T20:56:35.577802Z","iopub.status.idle":"2024-10-03T20:56:35.602824Z","shell.execute_reply.started":"2024-10-03T20:56:35.577758Z","shell.execute_reply":"2024-10-03T20:56:35.601966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_pred)\nprint(f'\\nValidation Accuracy: {accuracy:.4f}')\nprint('\\nClassification Report:')\nprint(classification_report(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:56:58.842647Z","iopub.execute_input":"2024-10-03T20:56:58.843610Z","iopub.status.idle":"2024-10-03T20:56:58.864557Z","shell.execute_reply.started":"2024-10-03T20:56:58.843567Z","shell.execute_reply":"2024-10-03T20:56:58.863475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_val, y_pred)\nplt.figure(figsize=(8,6))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', \n            xticklabels=model.classes_, yticklabels=model.classes_)\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:58:08.258634Z","iopub.execute_input":"2024-10-03T20:58:08.259780Z","iopub.status.idle":"2024-10-03T20:58:08.578929Z","shell.execute_reply.started":"2024-10-03T20:58:08.259724Z","shell.execute_reply":"2024-10-03T20:58:08.577962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = model.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:58:44.320935Z","iopub.execute_input":"2024-10-03T20:58:44.322065Z","iopub.status.idle":"2024-10-03T20:58:44.343235Z","shell.execute_reply.started":"2024-10-03T20:58:44.322012Z","shell.execute_reply":"2024-10-03T20:58:44.342031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if target_col in sample_submission.columns:\n    sample_submission[target_col] = test_predictions\nelse:\n    sample_submission['sii'] = test_predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:59:18.676331Z","iopub.execute_input":"2024-10-03T20:59:18.677024Z","iopub.status.idle":"2024-10-03T20:59:18.682567Z","shell.execute_reply.started":"2024-10-03T20:59:18.676977Z","shell.execute_reply":"2024-10-03T20:59:18.681554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index=False)\nprint(\"\\nSubmission file 'submission.csv' created successfully.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T20:59:43.140770Z","iopub.execute_input":"2024-10-03T20:59:43.141262Z","iopub.status.idle":"2024-10-03T20:59:43.149498Z","shell.execute_reply.started":"2024-10-03T20:59:43.141217Z","shell.execute_reply":"2024-10-03T20:59:43.148229Z"},"trusted":true},"execution_count":null,"outputs":[]}]}