{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9560605,"sourceType":"datasetVersion","datasetId":5826116}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler, PolynomialFeatures,FunctionTransformer, LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA\nfrom sklearn.metrics import cohen_kappa_score\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport warnings\nfrom sklearn.impute import KNNImputer\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom datetime import datetime\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-20T21:07:19.067207Z","iopub.execute_input":"2024-10-20T21:07:19.067786Z","iopub.status.idle":"2024-10-20T21:07:19.076809Z","shell.execute_reply.started":"2024-10-20T21:07:19.067735Z","shell.execute_reply":"2024-10-20T21:07:19.075389Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define helper functions to load and process time series data\ndef time_features(df):\n    \"\"\"\n    Extract statistical features from the time series data.\n    \"\"\"\n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    \n    # Define conditions for night, day, and no mask (full data)\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n    \n    # Define weekend and last week conditions\n    weekend = (df[\"weekday\"] >= 6)\n    last_week = df[\"relative_date_PCIAT\"] >= (df[\"relative_date_PCIAT\"].max() - 7)\n    \n    # Create additional weekend features\n    df[\"enmo_weekend\"] = df[\"enmo\"].where(weekend)\n    df[\"anglez_weekend\"] = df[\"anglez\"].where(weekend)\n    \n    # Basic features\n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"battery_voltage\"].mean(),\n        df[\"battery_voltage\"].diff().mean(),\n        df[\"relative_date_PCIAT\"].tail(1).values[0]\n    ]\n    \n    # List of columns of interest and masks\n    keys = [\"enmo\", \"anglez\", \"light\", \"enmo_weekend\", \"anglez_weekend\"]\n    masks = [no_mask, night, day, last_week]\n    \n    # Helper function for feature extraction\n    def extract_stats(data):\n        return [\n            data.mean(),\n            data.std(),\n            data.max(),\n            data.min(),\n            data.kurtosis(),\n            data.skew(),\n            data.diff().mean(),\n            data.diff().std(),\n            data.diff().quantile(0.9),\n            data.diff().quantile(0.1)\n        ]\n    \n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n    \n    return features\n\ndef process_file(filename, dirname):\n    \"\"\"\n    Process each parquet file and extract time features.\n    \"\"\"\n    # Read parquet file\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return time_features(df), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    \"\"\"\n    Load time series data from directory in parallel.\n    \"\"\"\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ndef recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan","metadata":{"execution":{"iopub.status.busy":"2024-10-20T21:07:19.080891Z","iopub.execute_input":"2024-10-20T21:07:19.081299Z","iopub.status.idle":"2024-10-20T21:07:19.105959Z","shell.execute_reply.started":"2024-10-20T21:07:19.081256Z","shell.execute_reply":"2024-10-20T21:07:19.104763Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Uncomment the following lines if you wish to include time series features\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Merge time series features if loaded\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n# Drop 'id' from training and test sets as it's only needed for submission\ntrain = train.drop('id', axis=1)\ntest_ids = test['id']  # Preserve test ids for submission\ntest = test.drop('id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-20T21:07:19.108678Z","iopub.execute_input":"2024-10-20T21:07:19.109641Z","iopub.status.idle":"2024-10-20T21:11:36.532438Z","shell.execute_reply.started":"2024-10-20T21:07:19.109580Z","shell.execute_reply":"2024-10-20T21:11:36.530935Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Recalculate SII based on PCIAT_Total and handle missing values\nPCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n# Apply the recalculate_sii function\ntrain['sii'] = train.apply(recalculate_sii, axis=1)\n\n# Drop rows with NaN in 'sii' as they cannot be used for training\ntrain = train.dropna(subset=['sii'])\n\n# Ensure 'sii' is integer type\ntrain['sii'] = train['sii'].astype(int)\n\n# Define feature columns\nkeepcols = ['Basic_Demos-Age', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday'] + ['stat_' + str(i) for i in range(204)]\n\n# Features and target\ntarget = train['sii']\nfeatures = train[keepcols]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-20T21:11:36.534186Z","iopub.execute_input":"2024-10-20T21:11:36.534592Z","iopub.status.idle":"2024-10-20T21:11:38.397939Z","shell.execute_reply.started":"2024-10-20T21:11:36.534546Z","shell.execute_reply":"2024-10-20T21:11:38.396784Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Initialize KNNImputer\nimputer = KNNImputer(n_neighbors=35)\n\n# Fit and transform the DataFrame\ndf_imputed = imputer.fit_transform(features)\ntrain_imputed = pd.DataFrame(df_imputed, columns=features.columns)\ntest_features = test[keepcols]\ntest_imputed = imputer.transform(test_features)\n\n\n\n# Initialize and train the model\nmodel = HistGradientBoostingClassifier()\nmodel.fit(train_imputed, target)\n\n# Make predictions on the test set\npredictions = model.predict(test_imputed)\n\n# Ensure predictions are within the valid range [0, 1, 2, 3]\npredictions = np.clip(predictions, 0, 3).astype(int)\n\n# Prepare the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'sii': predictions\n})\n\n# Verify the submission format\nprint(submission.head())\n\n# Save the submission to a CSV file\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-20T21:11:38.400205Z","iopub.execute_input":"2024-10-20T21:11:38.400591Z","iopub.status.idle":"2024-10-20T21:11:58.121509Z","shell.execute_reply.started":"2024-10-20T21:11:38.400550Z","shell.execute_reply":"2024-10-20T21:11:58.120571Z"},"trusted":true},"outputs":[],"execution_count":null}]}