{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T11:26:04.882288Z","iopub.execute_input":"2024-12-05T11:26:04.883298Z","iopub.status.idle":"2024-12-05T11:26:06.511789Z","shell.execute_reply.started":"2024-12-05T11:26:04.883257Z","shell.execute_reply":"2024-12-05T11:26:06.510495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import KFold\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tqdm import tqdm\nimport numpy as np\n\n# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Handle non-numeric data by converting categorical columns\nX = train.drop(['sii'], axis=1)  # Features\ny = train['sii']  # Target\n\n# Convert categorical columns to numeric using pd.get_dummies (one-hot encoding)\nX = pd.get_dummies(X)\ntest = pd.get_dummies(test)\n\n# Align test columns with train\nmissing_cols = set(X.columns) - set(test.columns)\nif missing_cols:\n    # Convert set to list for DataFrame columns\n    missing_df = pd.DataFrame(0, index=test.index, columns=list(missing_cols))\n    test = pd.concat([test, missing_df], axis=1)\n\n# Ensure test has the same columns as train\ntest = test[X.columns]\n\n# Neural Network model definition\ndef build_nn_model(input_dim):\n    model = Sequential([\n        Input(shape=(input_dim,)),\n        Dense(128, activation='relu'),\n        Dropout(0.2),\n        Dense(64, activation='relu'),\n        Dropout(0.2),\n        Dense(32, activation='relu'),\n        Dense(1, activation='linear')  # Linear activation for regression task\n    ])\n    model.compile(optimizer=Adam(), loss='mean_squared_error')\n    return model\n\n# Model training function\ndef TrainML(model_class, test_data, X, y, n_splits=5):\n    scaler = StandardScaler()\n    X = scaler.fit_transform(X)\n    test_data = scaler.transform(test_data)\n\n    oof_preds = np.zeros(len(y))\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    for fold, (train_idx, val_idx) in enumerate(tqdm(kf.split(X, y), desc=\"Training Folds\")):\n        X_train, X_val = X[train_idx], X[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        model = build_nn_model(X_train.shape[1])\n        model.fit(X_train, y_train, epochs=50, batch_size=32, verbose=0)\n\n        # Out-of-Fold Predictions\n        oof_preds[val_idx] = model.predict(X_val).flatten()\n\n        # Test Predictions for this fold\n        test_preds[:, fold] = model.predict(test_data).flatten()\n\n    # Average test predictions across folds\n    return oof_preds, np.mean(test_preds, axis=1)\n\n# Train the model and get predictions\noof_preds, test_preds_avg = TrainML(build_nn_model, test, X, y)\n\n# Handle NaNs or infinite values in predictions before rounding\ntest_preds_avg = np.nan_to_num(test_preds_avg, nan=0.0)\n\n# Prepare submission\nsample['sii'] = np.round(test_preds_avg).astype(int)\nsample.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T11:26:06.513926Z","iopub.execute_input":"2024-12-05T11:26:06.514545Z","iopub.status.idle":"2024-12-05T11:28:18.500019Z","shell.execute_reply.started":"2024-12-05T11:26:06.514493Z","shell.execute_reply":"2024-12-05T11:28:18.498782Z"}},"outputs":[],"execution_count":null}]}