{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.optimize import minimize\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Conv1D, Flatten, MaxPooling1D\nimport torch\n\n# Seed for reproducibility\nSEED = 42\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\n# Load and preprocess data\ndef load_data():\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    \n    train = train.dropna(subset=['sii'])  # Keep labeled values only\n    season_mapping = {'Winter': -1, 'Spring': -0.5, 'Summer': 0.5, 'Fall': 1}\n    train.replace(season_mapping, inplace=True)\n    test.replace(season_mapping, inplace=True)\n\n    return train, test\n\n# CNN model for numeric features\ndef build_cnn(input_shape):\n    model = Sequential([\n        Conv1D(filters=64, kernel_size=3, activation='relu', input_shape=input_shape),\n        MaxPooling1D(pool_size=2),\n        BatchNormalization(),\n        Dropout(0.3),\n        Conv1D(filters=128, kernel_size=3, activation='relu'),\n        MaxPooling1D(pool_size=2),\n        BatchNormalization(),\n        Dropout(0.3),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='linear')  # Regression output\n    ])\n    model.compile(optimizer='adam', loss='mse', metrics=['mae'])\n    return model\n\n# Custom threshold rounding\ndef threshold_rounder(preds, thresholds):\n    return np.where(preds < thresholds[0], 0,\n                    np.where(preds < thresholds[1], 1,\n                             np.where(preds < thresholds[2], 2, 3)))\n\n# Optimize thresholds\ndef optimize_thresholds(y_true, preds):\n    def evaluate_predictions(thresholds):\n        rounded_preds = threshold_rounder(preds, thresholds)\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n    opt_result = minimize(evaluate_predictions, [0.5, 1.5, 2.5], method='Nelder-Mead')\n    return opt_result.x\n\n#Train and Predict\ndef train_and_predict(train, test):\n    imputer = KNNImputer(n_neighbors=4)\n    scaler = StandardScaler()\n\n    X = train.drop(columns=['id', 'sii'])\n    y = train['sii']\n\n    test_ids = test['id']\n    X_test = test.drop(columns=['id'])\n\n    # Align test dataset to have the same columns as train\n    missing_cols = set(X.columns) - set(X_test.columns)\n    for col in missing_cols:\n        X_test[col] = np.nan\n\n    # Reorder test columns to match train\n    X_test = X_test[X.columns]\n\n    X_imputed = imputer.fit_transform(X)\n    X_test_imputed = imputer.transform(X_test)\n\n    X_scaled = scaler.fit_transform(X_imputed)\n    X_test_scaled = scaler.transform(X_test_imputed)\n\n    n_splits = 5\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    test_preds = np.zeros((len(X_test), n_splits))\n    oof_preds = np.zeros(len(y))\n    qwk_scores = []\n\n    for fold, (train_idx, val_idx) in enumerate(skf.split(X_scaled, y)):\n        X_train, X_val = X_scaled[train_idx], X_scaled[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        # Reshape for Conv1D\n        X_train_cnn = X_train.reshape(-1, X_train.shape[1], 1)\n        X_val_cnn = X_val.reshape(-1, X_val.shape[1], 1)\n        X_test_cnn = X_test_scaled.reshape(-1, X_test_scaled.shape[1], 1)\n\n        # Train CNN\n        cnn = build_cnn(input_shape=(X_train.shape[1], 1))\n        cnn.fit(X_train_cnn, y_train, validation_data=(X_val_cnn, y_val),\n                epochs=10, batch_size=32, verbose=1)\n\n        # Predict\n        cnn_val_preds = cnn.predict(X_val_cnn).flatten()\n        cnn_test_preds = cnn.predict(X_test_cnn).flatten()\n\n        oof_preds[val_idx] = cnn_val_preds\n        test_preds[:, fold] = cnn_test_preds\n\n        # Calculate QWK for this fold\n        val_qwk = cohen_kappa_score(y_val, threshold_rounder(cnn_val_preds, [0.5, 1.5, 2.5]), weights='quadratic')\n        qwk_scores.append(val_qwk)\n        print(f\"Fold {fold + 1} - Validation QWK: {val_qwk:.4f}\")\n\n    # Calculate OOF QWK\n    optimal_thresholds = optimize_thresholds(y, oof_preds)\n    oof_qwk = cohen_kappa_score(y, threshold_rounder(oof_preds, optimal_thresholds), weights='quadratic')\n    print(f\"Out-of-Fold QWK: {oof_qwk:.4f}\")\n\n    # Final Test Predictions\n    final_test_preds = test_preds.mean(axis=1)\n    final_test_preds = threshold_rounder(final_test_preds, optimal_thresholds)\n\n    submission = pd.DataFrame({'id': test_ids, 'sii': final_test_preds})\n    submission.to_csv('/kaggle/working/submission.csv', index=False)\n    print(submission)\n\nif __name__ == \"__main__\":\n    train_data, test_data = load_data()\n    train_and_predict(train_data, test_data)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-12T12:38:35.344192Z","iopub.execute_input":"2024-12-12T12:38:35.344630Z","iopub.status.idle":"2024-12-12T12:39:44.397888Z","shell.execute_reply.started":"2024-12-12T12:38:35.344594Z","shell.execute_reply":"2024-12-12T12:39:44.396996Z"}},"outputs":[],"execution_count":null}]}