{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import required libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:13.447678Z","iopub.execute_input":"2024-12-04T16:28:13.447952Z","iopub.status.idle":"2024-12-04T16:28:14.348979Z","shell.execute_reply.started":"2024-12-04T16:28:13.447911Z","shell.execute_reply":"2024-12-04T16:28:14.347930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.350088Z","iopub.execute_input":"2024-12-04T16:28:14.350458Z","iopub.status.idle":"2024-12-04T16:28:14.355842Z","shell.execute_reply.started":"2024-12-04T16:28:14.350429Z","shell.execute_reply":"2024-12-04T16:28:14.354837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Read the train CSV file","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\", index_col=\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.358408Z","iopub.execute_input":"2024-12-04T16:28:14.358747Z","iopub.status.idle":"2024-12-04T16:28:14.416006Z","shell.execute_reply.started":"2024-12-04T16:28:14.358694Z","shell.execute_reply":"2024-12-04T16:28:14.414881Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dataset overview","metadata":{}},{"cell_type":"markdown","source":"Delete row which not have output","metadata":{}},{"cell_type":"code","source":"df = df.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.417431Z","iopub.execute_input":"2024-12-04T16:28:14.417873Z","iopub.status.idle":"2024-12-04T16:28:14.427096Z","shell.execute_reply.started":"2024-12-04T16:28:14.417827Z","shell.execute_reply":"2024-12-04T16:28:14.426109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Remove columns which not appear in test.csv","metadata":{}},{"cell_type":"markdown","source":"One-hot encoding process","metadata":{}},{"cell_type":"code","source":"#Helper function\ndef encode_and_bind(original_dataframe, feature_to_encode):\n    dummies = pd.get_dummies(original_dataframe[[feature_to_encode]], dtype=int)\n    original_dataframe = pd.concat([original_dataframe, dummies], axis=1)\n    original_dataframe = original_dataframe.drop([feature_to_encode], axis=1)\n    return original_dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.428293Z","iopub.execute_input":"2024-12-04T16:28:14.428627Z","iopub.status.idle":"2024-12-04T16:28:14.439663Z","shell.execute_reply.started":"2024-12-04T16:28:14.428599Z","shell.execute_reply":"2024-12-04T16:28:14.438638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Proceed with encoding\ncategorical_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n       'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n       'PAQ_C-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ndf_encoded = df\nfor col in categorical_cols:\n    df_encoded = encode_and_bind(df_encoded, col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.441174Z","iopub.execute_input":"2024-12-04T16:28:14.441605Z","iopub.status.idle":"2024-12-04T16:28:14.510534Z","shell.execute_reply.started":"2024-12-04T16:28:14.441557Z","shell.execute_reply":"2024-12-04T16:28:14.509542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"binary_columns = [col for col in df_encoded.columns if df_encoded[col].nunique() == 2]\ncolumns_to_standardize = [col for col in df_encoded.columns if col not in binary_columns and col != 'sii']\ndf_encoded[columns_to_standardize] = scaler.fit_transform(df_encoded[columns_to_standardize])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.511689Z","iopub.execute_input":"2024-12-04T16:28:14.512094Z","iopub.status.idle":"2024-12-04T16:28:14.549752Z","shell.execute_reply.started":"2024-12-04T16:28:14.512048Z","shell.execute_reply":"2024-12-04T16:28:14.548591Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Bring output to end","metadata":{}},{"cell_type":"code","source":"cols = [col for col in df_encoded.columns if col != 'sii'] + ['sii']\ndf_encoded = df_encoded[cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.551175Z","iopub.execute_input":"2024-12-04T16:28:14.551571Z","iopub.status.idle":"2024-12-04T16:28:14.562200Z","shell.execute_reply.started":"2024-12-04T16:28:14.551534Z","shell.execute_reply":"2024-12-04T16:28:14.561140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#missing colums\nmissing_columns = ['PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04',\n       'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08',\n       'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n       'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16',\n       'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20',\n       'PCIAT-PCIAT_Total']\ndf_processed = df_encoded\nfor column in missing_columns:\n    df_processed = df_processed.drop(columns=column)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.563432Z","iopub.execute_input":"2024-12-04T16:28:14.563757Z","iopub.status.idle":"2024-12-04T16:28:14.616970Z","shell.execute_reply.started":"2024-12-04T16:28:14.563726Z","shell.execute_reply":"2024-12-04T16:28:14.615989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = [col for col in df_processed.columns if col != 'sii']\nX = df_processed[features]\ny = df_processed.sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.618360Z","iopub.execute_input":"2024-12-04T16:28:14.618770Z","iopub.status.idle":"2024-12-04T16:28:14.627893Z","shell.execute_reply.started":"2024-12-04T16:28:14.618722Z","shell.execute_reply":"2024-12-04T16:28:14.626906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Fill missing cells\nX = X.fillna(X.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.629154Z","iopub.execute_input":"2024-12-04T16:28:14.629568Z","iopub.status.idle":"2024-12-04T16:28:14.670303Z","shell.execute_reply.started":"2024-12-04T16:28:14.629528Z","shell.execute_reply":"2024-12-04T16:28:14.669369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exclusion_list = [\n    \"Fitness_Endurance-Season_Winter\", \"BIA-Season_Spring\", \n    \"PAQ_A-Season_Fall\", \"PAQ_A-Season_Spring\", \"PAQ_A-Season_Winter\", \n    \"PCIAT-Season_Fall\", \"PCIAT-Season_Spring\", \"PCIAT-Season_Summer\", \n    \"PCIAT-Season_Winter\"\n]\nfiltered_features = [feature for feature in features if feature not in exclusion_list]\nX = X[filtered_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.673925Z","iopub.execute_input":"2024-12-04T16:28:14.674344Z","iopub.status.idle":"2024-12-04T16:28:14.684902Z","shell.execute_reply.started":"2024-12-04T16:28:14.674301Z","shell.execute_reply":"2024-12-04T16:28:14.684005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:28:14.685960Z","iopub.execute_input":"2024-12-04T16:28:14.686226Z","iopub.status.idle":"2024-12-04T16:28:14.720484Z","shell.execute_reply.started":"2024-12-04T16:28:14.686200Z","shell.execute_reply":"2024-12-04T16:28:14.719512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrf_model = RandomForestClassifier(\n    n_estimators=200,\n    max_depth=10,\n    max_features='sqrt',\n    min_samples_split=5,\n    min_samples_leaf=2,\n    class_weight='balanced',\n    random_state=42,\n)\n\nrf_model.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:34:02.182845Z","iopub.execute_input":"2024-12-04T16:34:02.183190Z","iopub.status.idle":"2024-12-04T16:34:03.481015Z","shell.execute_reply.started":"2024-12-04T16:34:02.183161Z","shell.execute_reply":"2024-12-04T16:34:03.480095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nval_model = RandomForestClassifier(\n    n_estimators=100,\n    max_depth=20,\n    max_features='sqrt',\n    min_samples_split=2,\n    min_samples_leaf=4,\n    class_weight='balanced',\n    random_state=42,\n)\n\nval_model.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:55:07.638888Z","iopub.execute_input":"2024-12-04T16:55:07.639268Z","iopub.status.idle":"2024-12-04T16:55:08.368531Z","shell.execute_reply.started":"2024-12-04T16:55:07.639232Z","shell.execute_reply":"2024-12-04T16:55:08.367514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluation function","metadata":{}},{"cell_type":"code","source":"#Evaluation function\ndef quadratic_weighted_kappa(y_true, y_pred, n_classes):\n    \"\"\"\n    Calculate the Quadratic Weighted Kappa (QWK) score.\n\n    Parameters:\n    y_true (list or numpy array): Actual values (ground truth).\n    y_pred (list or numpy array): Predicted values.\n    n_classes (int): Number of distinct classes/labels.\n\n    Returns:\n    float: QWK score.\n    \"\"\"\n    # Create histogram matrix O (observed matrix)\n    O = np.zeros((n_classes, n_classes), dtype=np.float64)\n    for true, pred in zip(y_true, y_pred):\n        O[true, pred] += 1\n\n    # Create weight matrix W\n    W = np.zeros((n_classes, n_classes), dtype=np.float64)\n    for i in range(n_classes):\n        for j in range(n_classes):\n            W[i, j] = ((i - j) ** 2) / ((n_classes - 1) ** 2)\n\n    # Create expected matrix E\n    actual_hist = np.sum(O, axis=1)\n    pred_hist = np.sum(O, axis=0)\n    E = np.outer(actual_hist, pred_hist) / np.sum(O)\n\n    # Calculate QWK\n    numerator = np.sum(W * O)\n    denominator = np.sum(W * E)\n    kappa = 1 - (numerator / denominator)\n\n    return kappa\n\nval_preds = val_model.predict(X_val)\nval_preds = np.array(val_preds).astype(int)\n\ny_val = np.array(y_val).astype(int)\n\nprint(quadratic_weighted_kappa(y_val, val_preds, 4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:55:13.090422Z","iopub.execute_input":"2024-12-04T16:55:13.091251Z","iopub.status.idle":"2024-12-04T16:55:13.120701Z","shell.execute_reply.started":"2024-12-04T16:55:13.091215Z","shell.execute_reply":"2024-12-04T16:55:13.119824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Handle test data","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:32:41.022224Z","iopub.execute_input":"2024-12-04T16:32:41.022627Z","iopub.status.idle":"2024-12-04T16:32:41.032885Z","shell.execute_reply.started":"2024-12-04T16:32:41.022595Z","shell.execute_reply":"2024-12-04T16:32:41.031750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_categorical_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n       'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n       'PAQ_C-Season','SDS-Season', 'PreInt_EduHx-Season']\nfor col in test_categorical_cols:\n    test_data = encode_and_bind(test_data, col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:32:43.030397Z","iopub.execute_input":"2024-12-04T16:32:43.031279Z","iopub.status.idle":"2024-12-04T16:32:43.068091Z","shell.execute_reply.started":"2024-12-04T16:32:43.031243Z","shell.execute_reply":"2024-12-04T16:32:43.067110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"binary_columns = [col for col in test_data.columns if test_data[col].nunique() == 2]\ncolumns_to_standardize = [col for col in test_data.columns if col not in binary_columns and col != 'sii' and col != 'id']\ntest_data[columns_to_standardize] = scaler.fit_transform(test_data[columns_to_standardize])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:32:45.017128Z","iopub.execute_input":"2024-12-04T16:32:45.017975Z","iopub.status.idle":"2024-12-04T16:32:45.044515Z","shell.execute_reply.started":"2024-12-04T16:32:45.017937Z","shell.execute_reply":"2024-12-04T16:32:45.043392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_X = test_data[filtered_features]\n\ntest_X = test_X.fillna(test_X.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:32:47.530085Z","iopub.execute_input":"2024-12-04T16:32:47.530808Z","iopub.status.idle":"2024-12-04T16:32:47.566039Z","shell.execute_reply.started":"2024-12-04T16:32:47.530755Z","shell.execute_reply":"2024-12-04T16:32:47.565195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:36:43.416675Z","iopub.execute_input":"2024-12-01T09:36:43.417049Z","iopub.status.idle":"2024-12-01T09:36:43.449297Z","shell.execute_reply.started":"2024-12-01T09:36:43.417019Z","shell.execute_reply":"2024-12-01T09:36:43.447979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds = val_model.predict(test_X)\ntest_preds = np.array(test_preds).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:47:00.350627Z","iopub.execute_input":"2024-12-04T16:47:00.351000Z","iopub.status.idle":"2024-12-04T16:47:00.369609Z","shell.execute_reply.started":"2024-12-04T16:47:00.350969Z","shell.execute_reply":"2024-12-04T16:47:00.368798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submit","metadata":{}},{"cell_type":"code","source":"# Run the code to save predictions in the format used for competition scoring\n\noutput = pd.DataFrame({'id': test_data.id,\n                       'sii': test_preds})\noutput.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T16:47:02.845589Z","iopub.execute_input":"2024-12-04T16:47:02.846393Z","iopub.status.idle":"2024-12-04T16:47:02.853238Z","shell.execute_reply.started":"2024-12-04T16:47:02.846340Z","shell.execute_reply":"2024-12-04T16:47:02.852187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}