{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, classification_report, cohen_kappa_score\nfrom sklearn.impute import KNNImputer\nimport pickle\n\ndef get_clean_data(data_path, reference_cols=None):\n    # Load the data\n    data = pd.read_csv(data_path)\n    \n    # Drop columns containing 'Season' in their names\n    season_cols = [col for col in data.columns if 'Season' in col]\n    data = data.drop(season_cols, axis=1)\n    \n    # Drop 'id' column if it exists\n    if 'id' in data.columns:\n        data = data.drop('id', axis=1)\n    \n    # Identify numeric columns for imputation\n    numeric_cols = data.select_dtypes(include=['float64', 'int64']).columns\n    \n    # Apply KNN Imputer to handle missing values in numeric columns\n    imputer = KNNImputer(n_neighbors=5)\n    imputed_data = imputer.fit_transform(data[numeric_cols])\n    \n    # Create a DataFrame from the imputed data\n    data_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n    \n    # Ensure 'sii' column is integer type if it exists\n    if 'sii' in data_imputed.columns:\n        data_imputed['sii'] = data_imputed['sii'].round().astype(int)\n    \n    # Add back non-numeric columns to the imputed DataFrame\n    for col in data.columns:\n        if col not in numeric_cols:\n            data_imputed[col] = data[col]\n    \n    # Ensure the columns match between train and test data\n    if reference_cols is not None:\n        data_imputed = data_imputed.reindex(columns=reference_cols, fill_value=0)\n    \n    return data_imputed\n\ndef create_model(data): \n    X = data.drop(['sii'], axis=1)\n    y = data['sii']\n    \n    # Store column names before scaling\n    feature_columns = X.columns\n    \n    # Scale the data\n    scaler = StandardScaler()\n    X = scaler.fit_transform(X)\n    \n    # Train the model on the entire dataset\n    model = LogisticRegression(max_iter=500, class_weight='balanced')\n    model.fit(X, y)\n    \n    # Validation metrics on train set for verification\n    y_pred = model.predict(X)\n    print(\"Accuracy:\", accuracy_score(y, y_pred))\n    print(\"Classification report:\\n\", classification_report(y, y_pred))\n    print(\"Quadratic Weighted Kappa (QWK):\", cohen_kappa_score(y, y_pred, weights='quadratic'))\n    \n    return model, scaler, feature_columns\n\ndef generate_submission(model, scaler, test_data_path, reference_cols):\n    # Load the test data\n    test_data = pd.read_csv(test_data_path)\n    test_features = get_clean_data(test_data_path, reference_cols=reference_cols)\n    \n    # Transform the test features\n    test_features_scaled = scaler.transform(test_features)\n\n    # Predict on the test data\n    predictions = model.predict(test_features_scaled)\n\n    # Prepare submission DataFrame\n    submission = pd.DataFrame({\n        'id': test_data['id'],  # Use original 'id' from test_data\n        'sii': predictions\n    })\n    \n    # Save the submission\n    submission.to_csv('submission.csv', index=False)\n    print(\"Submission file 'submission.csv' is ready.\")\n\ndef main():\n    # Paths to datasets in Kaggle environment\n    train_data_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\"\n    test_data_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\"\n    \n    # Get cleaned training data\n    train_data = get_clean_data(train_data_path)\n\n    # Train model and save scaler\n    model, scaler, reference_cols = create_model(train_data)\n\n    # Save the model and scaler using pickle\n    with open('model.pkl', 'wb') as f:\n        pickle.dump(model, f)\n    \n    with open('scaler.pkl', 'wb') as f:\n        pickle.dump(scaler, f)\n    \n    # Generate submission file with consistent columns\n    generate_submission(model, scaler, test_data_path, reference_cols)\n\nif __name__ == '__main__':\n    main()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T07:03:33.041933Z","iopub.execute_input":"2024-11-05T07:03:33.043148Z","iopub.status.idle":"2024-11-05T07:03:41.786201Z","shell.execute_reply.started":"2024-11-05T07:03:33.043086Z","shell.execute_reply":"2024-11-05T07:03:41.785111Z"},"trusted":true},"execution_count":null,"outputs":[]}]}