{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-28T18:00:27.024282Z","iopub.execute_input":"2024-09-28T18:00:27.024843Z","iopub.status.idle":"2024-09-28T18:00:29.747221Z","shell.execute_reply.started":"2024-09-28T18:00:27.024783Z","shell.execute_reply":"2024-09-28T18:00:29.746024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:29.749857Z","iopub.execute_input":"2024-09-28T18:00:29.751026Z","iopub.status.idle":"2024-09-28T18:00:34.544500Z","shell.execute_reply.started":"2024-09-28T18:00:29.750967Z","shell.execute_reply":"2024-09-28T18:00:34.543321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Load data\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:34.545862Z","iopub.execute_input":"2024-09-28T18:00:34.546662Z","iopub.status.idle":"2024-09-28T18:00:34.635834Z","shell.execute_reply.started":"2024-09-28T18:00:34.546610Z","shell.execute_reply":"2024-09-28T18:00:34.634598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare columns between train and test sets\ntrain_columns = set(train_df.columns)\ntest_columns = set(test_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:34.637356Z","iopub.execute_input":"2024-09-28T18:00:34.637823Z","iopub.status.idle":"2024-09-28T18:00:34.643682Z","shell.execute_reply.started":"2024-09-28T18:00:34.637771Z","shell.execute_reply":"2024-09-28T18:00:34.642460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find missing columns in the test set compared to the training set\nmissing_in_test = train_columns - test_columns\nmissing_in_train = test_columns - train_columns","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:34.647404Z","iopub.execute_input":"2024-09-28T18:00:34.647891Z","iopub.status.idle":"2024-09-28T18:00:34.653979Z","shell.execute_reply.started":"2024-09-28T18:00:34.647841Z","shell.execute_reply":"2024-09-28T18:00:34.652650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Columns in training but not in test: {missing_in_test}\")\nprint(f\"Columns in test but not in training: {missing_in_train}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:34.655524Z","iopub.execute_input":"2024-09-28T18:00:34.655954Z","iopub.status.idle":"2024-09-28T18:00:34.664006Z","shell.execute_reply.started":"2024-09-28T18:00:34.655907Z","shell.execute_reply":"2024-09-28T18:00:34.662850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for basic statistics and missing data in training data\ntrain_df.info()\ntrain_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:34.665619Z","iopub.execute_input":"2024-09-28T18:00:34.666055Z","iopub.status.idle":"2024-09-28T18:00:34.904609Z","shell.execute_reply.started":"2024-09-28T18:00:34.666006Z","shell.execute_reply":"2024-09-28T18:00:34.903540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for basic statistics and missing data in test data\ntest_df.info()\ntest_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:34.905962Z","iopub.execute_input":"2024-09-28T18:00:34.906302Z","iopub.status.idle":"2024-09-28T18:00:35.040811Z","shell.execute_reply.started":"2024-09-28T18:00:34.906268Z","shell.execute_reply":"2024-09-28T18:00:35.039620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Check the percentage of missing values for 'sii'\nmissing_sii = train_df['sii'].isnull().sum()\ntotal_rows = train_df.shape[0]\nmissing_sii_percentage = (missing_sii / total_rows) * 100\nprint(f\"Percentage of missing 'sii': {missing_sii_percentage:.2f}%\")\n\n# 2. Drop rows where 'sii' is missing (if it's a significant percentage)\ntrain_df_clean = train_df.dropna(subset=['sii'])\n\n# 3. Handle missing numeric columns by imputing with median\nnumeric_cols = train_df_clean.select_dtypes(include=['float', 'int']).columns\nfor col in numeric_cols:\n    train_df_clean[col].fillna(train_df_clean[col].median(), inplace=True)\n\n# 4. Handle missing categorical columns by imputing with mode\ncategorical_cols = train_df_clean.select_dtypes(include=['object']).columns\nfor col in categorical_cols:\n    train_df_clean[col].fillna(train_df_clean[col].mode()[0], inplace=True)\n\n# 5. Check how many columns still have missing values after imputation\nprint(\"Remaining missing values after imputation:\")\nprint(train_df_clean.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:35.042387Z","iopub.execute_input":"2024-09-28T18:00:35.042795Z","iopub.status.idle":"2024-09-28T18:00:35.154261Z","shell.execute_reply.started":"2024-09-28T18:00:35.042752Z","shell.execute_reply":"2024-09-28T18:00:35.142906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# 1. Plot the distribution of the target variable 'sii'\nplt.figure(figsize=(8, 5))\nsns.countplot(data=train_df_clean, x='sii')\nplt.title('Distribution of Severity Impairment Index (sii)')\nplt.show()\n\n# 2. Visualize the distribution of internet use hours\nplt.figure(figsize=(8, 5))\nsns.histplot(data=train_df_clean, x='PreInt_EduHx-computerinternet_hoursday', kde=True)\nplt.title('Distribution of Internet Use Hours')\nplt.show()\n\n\n# 4. Check for missing values in the training set\nmissing_values = train_df_clean.isnull().sum()\nprint(f\"Missing values in the training set:\\n{missing_values}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:35.155747Z","iopub.execute_input":"2024-09-28T18:00:35.156149Z","iopub.status.idle":"2024-09-28T18:00:36.225814Z","shell.execute_reply.started":"2024-09-28T18:00:35.156108Z","shell.execute_reply":"2024-09-28T18:00:36.224624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_clean","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.227110Z","iopub.execute_input":"2024-09-28T18:00:36.227764Z","iopub.status.idle":"2024-09-28T18:00:36.274778Z","shell.execute_reply.started":"2024-09-28T18:00:36.227721Z","shell.execute_reply":"2024-09-28T18:00:36.273664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the columns in train and test datasets\ntrain_columns = set(train_df_clean.columns)\ntest_columns = set(test_df.columns)\n\n# Identify columns that are in train but not in test\nmissing_in_test = train_columns - test_columns\n\n# Identify columns in the test set that are not in the train set (if any)\nmissing_in_train = test_columns - train_columns\n\nprint(f\"Columns in training but not in test:\\n{missing_in_test}\")\nprint(f\"Columns in test but not in training:\\n{missing_in_train}\")\n\n# Drop the columns in training that are not in test set (excluding target 'sii')\ncolumns_to_drop = list(missing_in_test - {'sii'})  # Exclude target column 'sii'\ntrain_df_clean = train_df_clean.drop(columns=columns_to_drop)\n\nprint(f\"Columns dropped from the training set: {columns_to_drop}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.276501Z","iopub.execute_input":"2024-09-28T18:00:36.276875Z","iopub.status.idle":"2024-09-28T18:00:36.286112Z","shell.execute_reply.started":"2024-09-28T18:00:36.276836Z","shell.execute_reply":"2024-09-28T18:00:36.284939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the identified columns from both train and test sets (keeping only relevant features)\nX_train_cleaned = train_df_clean.drop(columns=['sii', 'id'])  # Exclude target and ID columns\nX_test_cleaned = test_df.drop(columns=['id'])  # Exclude ID column\n\nprint(f\"Final training columns: {X_train_cleaned.columns}\")\nprint(f\"Final test columns: {X_test_cleaned.columns}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.287693Z","iopub.execute_input":"2024-09-28T18:00:36.288183Z","iopub.status.idle":"2024-09-28T18:00:36.303025Z","shell.execute_reply.started":"2024-09-28T18:00:36.288127Z","shell.execute_reply":"2024-09-28T18:00:36.301821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_cleaned.info()\nX_test_cleaned.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.307490Z","iopub.execute_input":"2024-09-28T18:00:36.307886Z","iopub.status.idle":"2024-09-28T18:00:36.337285Z","shell.execute_reply.started":"2024-09-28T18:00:36.307846Z","shell.execute_reply":"2024-09-28T18:00:36.336247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert object columns to match the data types in the training data\n# We'll cast 'Season' and other categorical features to integers or appropriate types\nfor col in X_test_cleaned.select_dtypes(include=['object']).columns:\n    X_test_cleaned[col] = X_test_cleaned[col].astype('category').cat.codes\n\n# Fill missing values in test data\n# For numerical columns, we can fill with median values from the training set\nfor col in X_test_cleaned.select_dtypes(include=['float', 'int']).columns:\n    X_test_cleaned[col].fillna(X_train_cleaned[col].median(), inplace=True)\n\n# For categorical columns, we can fill missing values with the mode (most frequent category)\nfor col in X_test_cleaned.select_dtypes(include=['category']).columns:\n    X_test_cleaned[col].fillna(X_test_cleaned[col].mode()[0], inplace=True)\n\n# Verify that missing values have been handled and types are consistent\nprint(X_test_cleaned.info())\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.338719Z","iopub.execute_input":"2024-09-28T18:00:36.339140Z","iopub.status.idle":"2024-09-28T18:00:36.397771Z","shell.execute_reply.started":"2024-09-28T18:00:36.339092Z","shell.execute_reply":"2024-09-28T18:00:36.396669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_clean","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.399245Z","iopub.execute_input":"2024-09-28T18:00:36.399622Z","iopub.status.idle":"2024-09-28T18:00:36.442507Z","shell.execute_reply.started":"2024-09-28T18:00:36.399561Z","shell.execute_reply":"2024-09-28T18:00:36.441279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Instantiate LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Select the categorical columns in your dataset\ncategorical_columns = train_df_clean.select_dtypes(include=['object']).columns\n\n# Apply LabelEncoder to each categorical column\nfor col in categorical_columns:\n    train_df_clean[col] = label_encoder.fit_transform(train_df_clean[col])\n\n# Display the updated dataframe\ntrain_df_clean.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.444121Z","iopub.execute_input":"2024-09-28T18:00:36.444919Z","iopub.status.idle":"2024-09-28T18:00:36.498674Z","shell.execute_reply.started":"2024-09-28T18:00:36.444878Z","shell.execute_reply":"2024-09-28T18:00:36.497512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n\n\n# 2. Split data into features and target\nX = train_df_clean.drop(columns=['sii', 'id'])  # Drop target and ID columns\ny = train_df_clean['sii']\n\n# 3. Train-test split (for validation)\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, stratify=y, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:00:36.499998Z","iopub.execute_input":"2024-09-28T18:00:36.500339Z","iopub.status.idle":"2024-09-28T18:00:36.516306Z","shell.execute_reply.started":"2024-09-28T18:00:36.500303Z","shell.execute_reply":"2024-09-28T18:00:36.515138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import cohen_kappa_score\nimport numpy as np\n\n# Assuming X_train, X_val, y_train, y_val are already prepared\n# Make sure to scale the data\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\n# Define the MLP model\ndef create_mlp(input_shape):\n    model = Sequential()\n    \n    # First layer (input)\n    model.add(Dense(128, activation='relu', input_shape=(input_shape,)))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.3))\n    \n    # Second layer\n    model.add(Dense(64, activation='relu'))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.3))\n    \n    # Third layer\n    model.add(Dense(32, activation='relu'))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n    \n    # Output layer (since we have 4 classes, we use softmax)\n    model.add(Dense(4, activation='softmax'))\n    \n    return model\n\n# Initialize the model\ninput_shape = X_train_scaled.shape[1]\nmlp_model = create_mlp(input_shape)\n# Compile the model\nmlp_model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = mlp_model.fit(\n    X_train_scaled, y_train, \n    validation_data=(X_val_scaled, y_val),\n    epochs=50, \n    batch_size=64, \n    verbose=1\n)\n# Predict on validation data\ny_pred_probs = mlp_model.predict(X_val_scaled)\ny_pred = np.argmax(y_pred_probs, axis=1)\n\n# Calculate Quadratic Weighted Kappa\nqwk_score = cohen_kappa_score(y_val, y_pred, weights='quadratic')\nprint(f\"Quadratic Weighted Kappa: {qwk_score}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:09:48.286971Z","iopub.execute_input":"2024-09-28T18:09:48.287796Z","iopub.status.idle":"2024-09-28T18:10:09.541000Z","shell.execute_reply.started":"2024-09-28T18:09:48.287751Z","shell.execute_reply":"2024-09-28T18:10:09.539953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Scale the test data\nX_test_scaled = scaler.transform(X_test_cleaned)\n\n# Predict on test data\ntest_pred_probs = mlp_model.predict(X_test_scaled)\ntest_pred = np.argmax(test_pred_probs, axis=1)\n\n\n\n# Prepare the submission\nsubmission = pd.DataFrame({\n    'id': test_df['id'],\n    'sii': test_pred \n})\n\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-28T18:04:40.338723Z","iopub.execute_input":"2024-09-28T18:04:40.339484Z","iopub.status.idle":"2024-09-28T18:04:40.619542Z","shell.execute_reply.started":"2024-09-28T18:04:40.339439Z","shell.execute_reply":"2024-09-28T18:04:40.618564Z"},"trusted":true},"execution_count":null,"outputs":[]}]}