{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:31.166676Z","iopub.execute_input":"2024-12-19T18:09:31.167101Z","iopub.status.idle":"2024-12-19T18:09:32.229829Z","shell.execute_reply.started":"2024-12-19T18:09:31.167050Z","shell.execute_reply":"2024-12-19T18:09:32.228687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\nimport os\nfrom concurrent.futures import ThreadPoolExecutor\nfrom sklearn.base import clone\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.231700Z","iopub.execute_input":"2024-12-19T18:09:32.232048Z","iopub.status.idle":"2024-12-19T18:09:32.238380Z","shell.execute_reply.started":"2024-12-19T18:09:32.232015Z","shell.execute_reply":"2024-12-19T18:09:32.236970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to compute quadratic weighted kappa\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.239937Z","iopub.execute_input":"2024-12-19T18:09:32.240318Z","iopub.status.idle":"2024-12-19T18:09:32.252633Z","shell.execute_reply.started":"2024-12-19T18:09:32.240272Z","shell.execute_reply":"2024-12-19T18:09:32.251373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Threshold rounding function\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.255215Z","iopub.execute_input":"2024-12-19T18:09:32.255604Z","iopub.status.idle":"2024-12-19T18:09:32.267907Z","shell.execute_reply.started":"2024-12-19T18:09:32.255568Z","shell.execute_reply":"2024-12-19T18:09:32.266665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluation function\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.269336Z","iopub.execute_input":"2024-12-19T18:09:32.269698Z","iopub.status.idle":"2024-12-19T18:09:32.281809Z","shell.execute_reply.started":"2024-12-19T18:09:32.269663Z","shell.execute_reply":"2024-12-19T18:09:32.280390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define function to process each file\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)  # Drop unnecessary column if present\n    return df.describe().values.reshape(-1), filename.split('=')[1]  # Return statistics and ID\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.283157Z","iopub.execute_input":"2024-12-19T18:09:32.283557Z","iopub.status.idle":"2024-12-19T18:09:32.297966Z","shell.execute_reply.started":"2024-12-19T18:09:32.283519Z","shell.execute_reply":"2024-12-19T18:09:32.296441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to load and process all time series files\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)  # Get list of file IDs in directory\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))  # Parallel processing\n\n    stats, indexes = zip(*results)  # Extract statistics and IDs from results\n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])  # Create DataFrame\n    df['id'] = indexes  # Add ID column\n\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.299338Z","iopub.execute_input":"2024-12-19T18:09:32.299760Z","iopub.status.idle":"2024-12-19T18:09:32.318552Z","shell.execute_reply.started":"2024-12-19T18:09:32.299714Z","shell.execute_reply":"2024-12-19T18:09:32.317550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load CSV data\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.319930Z","iopub.execute_input":"2024-12-19T18:09:32.320307Z","iopub.status.idle":"2024-12-19T18:09:32.403718Z","shell.execute_reply.started":"2024-12-19T18:09:32.320241Z","shell.execute_reply":"2024-12-19T18:09:32.402753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load time series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:09:32.405517Z","iopub.execute_input":"2024-12-19T18:09:32.405846Z","iopub.status.idle":"2024-12-19T18:10:59.570286Z","shell.execute_reply.started":"2024-12-19T18:09:32.405815Z","shell.execute_reply":"2024-12-19T18:10:59.569073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge time series with tabular data\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)  # Remove ID column after merging\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.574511Z","iopub.execute_input":"2024-12-19T18:10:59.574921Z","iopub.status.idle":"2024-12-19T18:10:59.598237Z","shell.execute_reply.started":"2024-12-19T18:10:59.574877Z","shell.execute_reply":"2024-12-19T18:10:59.597277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define feature columns\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols  # Add time series columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.599622Z","iopub.execute_input":"2024-12-19T18:10:59.599997Z","iopub.status.idle":"2024-12-19T18:10:59.608588Z","shell.execute_reply.started":"2024-12-19T18:10:59.599955Z","shell.execute_reply":"2024-12-19T18:10:59.607560Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter relevant columns and drop rows with missing target values\ntrain = train[featuresCols]\ntrain = train.dropna(subset=['sii'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.610026Z","iopub.execute_input":"2024-12-19T18:10:59.610554Z","iopub.status.idle":"2024-12-19T18:10:59.635803Z","shell.execute_reply.started":"2024-12-19T18:10:59.610507Z","shell.execute_reply":"2024-12-19T18:10:59.634704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define categorical columns\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n         'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.637061Z","iopub.execute_input":"2024-12-19T18:10:59.637405Z","iopub.status.idle":"2024-12-19T18:10:59.642566Z","shell.execute_reply.started":"2024-12-19T18:10:59.637373Z","shell.execute_reply":"2024-12-19T18:10:59.641382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to process categorical columns\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')  # Fill missing values\n        df[c] = df[c].astype('category')  # Convert to category type\n    return df\n\ntrain = update(train)\ntest = update(test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.643938Z","iopub.execute_input":"2024-12-19T18:10:59.644325Z","iopub.status.idle":"2024-12-19T18:10:59.678935Z","shell.execute_reply.started":"2024-12-19T18:10:59.644273Z","shell.execute_reply":"2024-12-19T18:10:59.677967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping categorical columns to integers\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\n# Print final shapes of train and test datasets\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.680219Z","iopub.execute_input":"2024-12-19T18:10:59.680613Z","iopub.status.idle":"2024-12-19T18:10:59.730706Z","shell.execute_reply.started":"2024-12-19T18:10:59.680580Z","shell.execute_reply":"2024-12-19T18:10:59.729779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.ensemble import RandomForestRegressor\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.731894Z","iopub.execute_input":"2024-12-19T18:10:59.732189Z","iopub.status.idle":"2024-12-19T18:10:59.736847Z","shell.execute_reply.started":"2024-12-19T18:10:59.732158Z","shell.execute_reply":"2024-12-19T18:10:59.735787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop(columns=['sii'])\ny = train['sii']\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\nmodel = HistGradientBoostingRegressor(max_iter=100, random_state=42)\nmodel.fit(X_train, y_train)\n\ny_pred = model.predict(X_val)\n\nmse = mean_squared_error(y_val, y_pred)\nqwk = cohen_kappa_score(y_val, y_pred.round(), weights='quadratic')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:10:59.738057Z","iopub.execute_input":"2024-12-19T18:10:59.738404Z","iopub.status.idle":"2024-12-19T18:11:01.732366Z","shell.execute_reply.started":"2024-12-19T18:10:59.738360Z","shell.execute_reply":"2024-12-19T18:11:01.731483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nqwk_scores = []\n\nfor train_idx, val_idx in tqdm(kf.split(X, y), total=5, desc=\"K-Fold Training\"):\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    # Huấn luyện mô hình\n    model = HistGradientBoostingRegressor(max_iter=100, random_state=42)\n    model.fit(X_train, y_train)\n    \n    # Dự đoán và tính QWK\n    y_pred = model.predict(X_val)\n    qwk = cohen_kappa_score(y_val, y_pred.round(), weights='quadratic')\n    qwk_scores.append(qwk)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:11:01.733323Z","iopub.execute_input":"2024-12-19T18:11:01.733629Z","iopub.status.idle":"2024-12-19T18:11:08.939590Z","shell.execute_reply.started":"2024-12-19T18:11:01.733598Z","shell.execute_reply":"2024-12-19T18:11:08.938519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.fillna(X.mean())\n\nparam_grid = {\n    'n_estimators': [50, 100, 200],\n    'max_depth': [5, 10, 20, None],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4],\n}\n\n# Randomized Search\nrandom_search = RandomizedSearchCV(\n    estimator=RandomForestRegressor(random_state=42),\n    param_distributions=param_grid,\n    n_iter=20,\n    cv=3,\n    scoring='neg_mean_squared_error',\n    verbose=2,\n    random_state=42,\n    n_jobs=-1\n)\n\nrandom_search.fit(X, y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:11:08.941012Z","iopub.execute_input":"2024-12-19T18:11:08.941346Z","iopub.status.idle":"2024-12-19T18:13:53.807882Z","shell.execute_reply.started":"2024-12-19T18:11:08.941313Z","shell.execute_reply":"2024-12-19T18:13:53.806656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:13:53.809857Z","iopub.execute_input":"2024-12-19T18:13:53.810339Z","iopub.status.idle":"2024-12-19T18:13:53.827716Z","shell.execute_reply.started":"2024-12-19T18:13:53.810286Z","shell.execute_reply":"2024-12-19T18:13:53.826590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test.drop(columns=['sii'], errors='ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:16:26.936832Z","iopub.execute_input":"2024-12-19T18:16:26.937805Z","iopub.status.idle":"2024-12-19T18:16:26.944705Z","shell.execute_reply.started":"2024-12-19T18:16:26.937760Z","shell.execute_reply":"2024-12-19T18:16:26.943525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dự đoán trên tập kiểm tra\ntest_predictions = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:16:30.143211Z","iopub.execute_input":"2024-12-19T18:16:30.143721Z","iopub.status.idle":"2024-12-19T18:16:30.160192Z","shell.execute_reply.started":"2024-12-19T18:16:30.143677Z","shell.execute_reply":"2024-12-19T18:16:30.158238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Làm tròn giá trị dự đoán\ntest_predictions_rounded = test_predictions.round().astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:16:34.445057Z","iopub.execute_input":"2024-12-19T18:16:34.445496Z","iopub.status.idle":"2024-12-19T18:16:34.450564Z","shell.execute_reply.started":"2024-12-19T18:16:34.445453Z","shell.execute_reply":"2024-12-19T18:16:34.449222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo file submission\nsubmission = pd.DataFrame({\n    'id': sample['id'],\n    'sii': test_predictions_rounded\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T18:16:43.838867Z","iopub.execute_input":"2024-12-19T18:16:43.839283Z","iopub.status.idle":"2024-12-19T18:16:43.848784Z","shell.execute_reply.started":"2024-12-19T18:16:43.839231Z","shell.execute_reply":"2024-12-19T18:16:43.847798Z"}},"outputs":[],"execution_count":null}]}