{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:04.935127Z","iopub.execute_input":"2024-12-06T15:27:04.935497Z","iopub.status.idle":"2024-12-06T15:27:08.625838Z","shell.execute_reply.started":"2024-12-06T15:27:04.935461Z","shell.execute_reply":"2024-12-06T15:27:08.624750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Input, LeakyReLU, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:08.627992Z","iopub.execute_input":"2024-12-06T15:27:08.628486Z","iopub.status.idle":"2024-12-06T15:27:23.716046Z","shell.execute_reply.started":"2024-12-06T15:27:08.628449Z","shell.execute_reply":"2024-12-06T15:27:23.714838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load training and test data\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.717717Z","iopub.execute_input":"2024-12-06T15:27:23.718522Z","iopub.status.idle":"2024-12-06T15:27:23.794109Z","shell.execute_reply.started":"2024-12-06T15:27:23.718471Z","shell.execute_reply":"2024-12-06T15:27:23.792894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove columns with more than 50% missing data\nthreshold = 0.5 * len(train_data)\ncolumns_with_data = train_data.columns[train_data.isnull().sum() < threshold]\ntrain_data = train_data[columns_with_data]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.795465Z","iopub.execute_input":"2024-12-06T15:27:23.795792Z","iopub.status.idle":"2024-12-06T15:27:23.820636Z","shell.execute_reply.started":"2024-12-06T15:27:23.795761Z","shell.execute_reply":"2024-12-06T15:27:23.819514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace missing values with 0\ntrain_data.fillna(0, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.823648Z","iopub.execute_input":"2024-12-06T15:27:23.824234Z","iopub.status.idle":"2024-12-06T15:27:23.836635Z","shell.execute_reply.started":"2024-12-06T15:27:23.824185Z","shell.execute_reply":"2024-12-06T15:27:23.835399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop rows where the target 'sii' is missing\ntrain_data_cleaned = train_data.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.838329Z","iopub.execute_input":"2024-12-06T15:27:23.838640Z","iopub.status.idle":"2024-12-06T15:27:23.857638Z","shell.execute_reply.started":"2024-12-06T15:27:23.838609Z","shell.execute_reply":"2024-12-06T15:27:23.856430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Columns that need to be encoded\nseason_cols = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n    'FGC-Season', 'BIA-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.859137Z","iopub.execute_input":"2024-12-06T15:27:23.859602Z","iopub.status.idle":"2024-12-06T15:27:23.870885Z","shell.execute_reply.started":"2024-12-06T15:27:23.859538Z","shell.execute_reply":"2024-12-06T15:27:23.869753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define mapping for the seasons\nseason_mapping = {'Spring': 0, 'Summer': 1, 'Fall': 2, 'Winter': 3}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.872287Z","iopub.execute_input":"2024-12-06T15:27:23.872660Z","iopub.status.idle":"2024-12-06T15:27:23.889031Z","shell.execute_reply.started":"2024-12-06T15:27:23.872627Z","shell.execute_reply":"2024-12-06T15:27:23.887508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categorical variables in training data\nfor col in season_cols:\n    if col in train_data_cleaned.columns:\n        train_data_cleaned[col] = train_data_cleaned[col].map(season_mapping).fillna(-1).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.891000Z","iopub.execute_input":"2024-12-06T15:27:23.891583Z","iopub.status.idle":"2024-12-06T15:27:23.915251Z","shell.execute_reply.started":"2024-12-06T15:27:23.891545Z","shell.execute_reply":"2024-12-06T15:27:23.914046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define features and target variable\ncommon_columns = train_data_cleaned.columns.intersection(test_data.columns)\nX = train_data_cleaned[common_columns]\ny = train_data_cleaned['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.917062Z","iopub.execute_input":"2024-12-06T15:27:23.917422Z","iopub.status.idle":"2024-12-06T15:27:23.927635Z","shell.execute_reply.started":"2024-12-06T15:27:23.917389Z","shell.execute_reply":"2024-12-06T15:27:23.926640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert categorical columns to numeric using LabelEncoder\nencoder = LabelEncoder()\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\nfor col in categorical_columns:\n    X.loc[:, col] = encoder.fit_transform(X[col].astype(str))  # Use .loc to modify the DataFrame correctly","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.929434Z","iopub.execute_input":"2024-12-06T15:27:23.929842Z","iopub.status.idle":"2024-12-06T15:27:23.950114Z","shell.execute_reply.started":"2024-12-06T15:27:23.929806Z","shell.execute_reply":"2024-12-06T15:27:23.948582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert categorical columns to numeric using LabelEncoder\nencoder = LabelEncoder()\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\nfor col in categorical_columns:\n    X.loc[:, col] = encoder.fit_transform(X[col].astype(str))  # Using .loc to modify safely","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.951560Z","iopub.execute_input":"2024-12-06T15:27:23.951891Z","iopub.status.idle":"2024-12-06T15:27:23.972093Z","shell.execute_reply.started":"2024-12-06T15:27:23.951861Z","shell.execute_reply":"2024-12-06T15:27:23.971052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop ID column if exists\nX = X.drop(columns=['id'], errors='ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.973598Z","iopub.execute_input":"2024-12-06T15:27:23.974014Z","iopub.status.idle":"2024-12-06T15:27:23.991099Z","shell.execute_reply.started":"2024-12-06T15:27:23.973977Z","shell.execute_reply":"2024-12-06T15:27:23.989532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standardize the features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:23.994457Z","iopub.execute_input":"2024-12-06T15:27:23.994822Z","iopub.status.idle":"2024-12-06T15:27:24.017603Z","shell.execute_reply.started":"2024-12-06T15:27:23.994790Z","shell.execute_reply":"2024-12-06T15:27:24.016353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply PCA to reduce dimensionality (95% variance retained)\npca = PCA(n_components=0.95)\nX_pca = pca.fit_transform(X_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:24.019149Z","iopub.execute_input":"2024-12-06T15:27:24.019513Z","iopub.status.idle":"2024-12-06T15:27:24.058896Z","shell.execute_reply.started":"2024-12-06T15:27:24.019477Z","shell.execute_reply":"2024-12-06T15:27:24.057880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and testing sets (with stratification)\nX_train, X_test, y_train, y_test = train_test_split(X_pca, y, test_size=0.25, stratify=y, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:24.060020Z","iopub.execute_input":"2024-12-06T15:27:24.062106Z","iopub.status.idle":"2024-12-06T15:27:24.079899Z","shell.execute_reply.started":"2024-12-06T15:27:24.062051Z","shell.execute_reply":"2024-12-06T15:27:24.078650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the distribution of the target variable in the splits\nprint(f\"Training set distribution:\\n{y_train.value_counts(normalize=True)}\")\nprint(f\"Test set distribution:\\n{y_test.value_counts(normalize=True)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:24.081515Z","iopub.execute_input":"2024-12-06T15:27:24.082244Z","iopub.status.idle":"2024-12-06T15:27:24.112429Z","shell.execute_reply.started":"2024-12-06T15:27:24.082194Z","shell.execute_reply":"2024-12-06T15:27:24.111376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build a more complex Neural Network model with batch normalization and LeakyReLU\nmodel = Sequential([\n    Input(shape=(X_train.shape[1],)),  # Input layer\n    Dense(256),  # More neurons in the first layer\n    BatchNormalization(),  # Batch normalization\n    LeakyReLU(negative_slope=0.2),  # LeakyReLU activation function with negative_slope\n    Dropout(0.3),\n    \n    Dense(128),\n    BatchNormalization(),\n    LeakyReLU(negative_slope=0.2),  # Updated argument\n    Dropout(0.3),\n    \n    Dense(64),\n    BatchNormalization(),\n    LeakyReLU(negative_slope=0.2),  # Updated argument\n    Dropout(0.2),\n    \n    Dense(32),\n    BatchNormalization(),\n    LeakyReLU(negative_slope=0.2),  # Updated argument\n    \n    Dense(1, activation='sigmoid')  # Sigmoid for binary classification\n])\n\n# Compile the model with Adam optimizer\nmodel.compile(optimizer=Adam(learning_rate=0.001), loss='binary_crossentropy', metrics=['accuracy'])\n\n# Check model summary\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:24.117159Z","iopub.execute_input":"2024-12-06T15:27:24.117577Z","iopub.status.idle":"2024-12-06T15:27:24.330124Z","shell.execute_reply.started":"2024-12-06T15:27:24.117534Z","shell.execute_reply":"2024-12-06T15:27:24.329095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up early stopping and learning rate reduction\nearly_stop = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\nlr_reduction = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=5, min_lr=0.00001)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:24.331631Z","iopub.execute_input":"2024-12-06T15:27:24.332467Z","iopub.status.idle":"2024-12-06T15:27:24.337904Z","shell.execute_reply.started":"2024-12-06T15:27:24.332415Z","shell.execute_reply":"2024-12-06T15:27:24.336892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model with more epochs and added callbacks\nhistory = model.fit(X_train, y_train, \n                    validation_data=(X_test, y_test),\n                    epochs=100, \n                    batch_size=32, \n                    callbacks=[early_stop, lr_reduction],\n                    verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:24.339389Z","iopub.execute_input":"2024-12-06T15:27:24.340434Z","iopub.status.idle":"2024-12-06T15:28:07.938104Z","shell.execute_reply.started":"2024-12-06T15:27:24.340385Z","shell.execute_reply":"2024-12-06T15:28:07.936583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot accuracy and loss during training\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Model Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n\nplt.figure(figsize=(12, 6))\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:28:07.940279Z","iopub.execute_input":"2024-12-06T15:28:07.940681Z","iopub.status.idle":"2024-12-06T15:28:08.771488Z","shell.execute_reply.started":"2024-12-06T15:28:07.940645Z","shell.execute_reply":"2024-12-06T15:28:08.770169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess the test data\ntest_data.fillna(0, inplace=True)\n\n# Encode categorical columns in test data\nfor col in season_cols:\n    if col in test_data.columns:\n        test_data[col] = test_data[col].map(season_mapping).fillna(-1).astype(int)\n\n# Prepare the test features\nX_test_data = test_data[common_columns].drop(columns=['id'], errors='ignore')\n\n# Standardize and apply PCA to the test data\nX_test_scaled = scaler.transform(X_test_data)\nX_test_pca = pca.transform(X_test_scaled)\n\n# Check the shape of the processed test data\nprint(f\"Shape of test data after PCA: {X_test_pca.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:28:08.773182Z","iopub.execute_input":"2024-12-06T15:28:08.773677Z","iopub.status.idle":"2024-12-06T15:28:08.796038Z","shell.execute_reply.started":"2024-12-06T15:28:08.773628Z","shell.execute_reply":"2024-12-06T15:28:08.794697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions using the trained model\npredictions = model.predict(X_test_pca)\n\n# Convert probabilities to binary outcomes (0 or 1)\npredictions = (predictions > 0.5).astype(int).flatten()\n\n# Create a DataFrame for submission\nsubmission = pd.DataFrame({\n    'id': test_data['id'],\n    'sii': predictions\n})\n\n# Save the submission to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\n# Check the first few rows of the submission\nprint(submission.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:28:08.799998Z","iopub.execute_input":"2024-12-06T15:28:08.800465Z","iopub.status.idle":"2024-12-06T15:28:09.003504Z","shell.execute_reply.started":"2024-12-06T15:28:08.800425Z","shell.execute_reply":"2024-12-06T15:28:09.002322Z"}},"outputs":[],"execution_count":null}]}