{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Submission 2 (Nick's Modfications)\n\n1. Dropped BIA as it's often considered to be unreliable\n\n2. Dropped seasons as features\n\n3. Removed one-hot encoding\n\n\n\nThis improved performance to 0.3657\n\n\n","metadata":{}},{"cell_type":"code","source":"run_cv = False          # set this to True if you want to run k-fold CV, set to False if not\n\nrun_output = True      # set this to True if you want to run the model on the test data to generate an output","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:27.862956Z","iopub.status.busy":"2024-10-09T15:27:27.861789Z","iopub.status.idle":"2024-10-09T15:27:27.868905Z","shell.execute_reply":"2024-10-09T15:27:27.867484Z","shell.execute_reply.started":"2024-10-09T15:27:27.862836Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ParquetAnalyzer Class Definition\n\n\n\nfrom concurrent.futures import ThreadPoolExecutor\n\nfrom typing import Optional\n\nimport os\n\n\n\nimport matplotlib.pyplot as plt\n\nimport numpy as np\n\nimport pandas as pd\n\nimport plotly.express as px\n\nfrom tqdm import tqdm\n\n\n\nclass ParquetAnalyzer:\n\n    \"\"\"\n\n    A ParquetAnalyzer object gives access to raw or engineered features from available parquet data.\n\n\n\n    Public methods and properties for an instance of ParquetAnalyzer named `pa` are as follows:\n\n        - pa.parquet_ids\n\n        - pa.get_raw_features(subject_id)\n\n        - pa.engineer_features_for_subject(subject_id)\n\n        - pa.engineer_features_for_subjects()\n\n    Refer to each public method or property's docstring for more information.\n\n    \"\"\"\n\n    \n\n    # --- PUBLIC METHODS --- (these provide no-fuss access to parquet data)\n\n    \n\n    def __init__(self, partition: str = \"train\"):\n\n        \"\"\"\n\n        Initialize an ParquetAnalyzer object.\n\n\n\n        Arguments:\n\n            - partition - Either \"train\" or \"test\", corresponding to series_train.parquet and series_test.parquet.\n\n        \"\"\"\n\n        # Take inventory of which subject ID's have parquet data, and the paths to that parquet data.\n\n        self._parquet_paths = {}\n\n        self._root = f\"/kaggle/input/child-mind-institute-problematic-internet-use/series_{partition}.parquet/\"\n\n        for id_subdir in os.listdir(self._root):\n\n            parquet_path = os.path.join(self._root, id_subdir, 'part-0.parquet')\n\n            parquet_id = self._path_to_subject_id(parquet_path)\n\n            self._parquet_paths[parquet_id] = parquet_path\n\n    \n\n    @property\n\n    def parquet_ids(self) -> list[str]:\n\n        \"\"\"\n\n        A list of subject ID's which have parquet data available.\n\n        An example of an ID for a subject is \"00115b9f\".\n\n        \"\"\"\n\n        return self._parquet_ids\n\n    \n\n    def get_raw_features(self, subject_id: str) -> pd.DataFrame:\n\n        \"\"\"\n\n        Return a pd.DataFrame of raw parquet data for a single subject.\n\n        \n\n        The following raw time signals are its columns (\"+\" indicates signals that are most likely to be useful):\n\n\n\n        Arguments:\n\n            - subject_id - The ID of a subject who has parquet data (such as \"00115b9f\")\n\n            \n\n        Returns:\n\n            - A table whose rows correspond to timesteps identified by the index \"step\", and whose columns are the following values at each timestep:\n\n                - X is a raw accelerometer signal captured in the X-axis. \n\n                    There may be direction-specific motion of interest, but for now assume ENMO summarizes these three, so don't necessarily need to use yet.\n\n                - Y is similar to X, but for the Y-axis.\n\n                - Z is similar to X, but for the Z-axis.\n\n                - enmo is approximately motion magnitude (Euclidean Norm Minus One). \n\n                    Could hypothesize this is low but possibly jittery (high frequency) for the chronically online, for example.\n\n                - anglez is angle from horizontal plane. \n\n                - non-wear_flag\n\n                - light is the amount of light sensed by the device. \n\n                    Probably corresponds to being indoors or outdoors - so a lower value might indicate being indoors. \n\n                    Even lower values might also suggest being indoors in the dark.\n\n                - battery_voltage \n\n                    Shouldn't matter, but lower battery might cause noise or attenuate signals?\n\n                - time_of_day could be used to partition data. \n\n                    If there's any motion late at night, it might imply the user is a night owl.\n\n                - weekday (i.e. which day of the week it is - 6 or 7 would be the weekend)\n\n                    Could also be interesting to partition on.\n\n                - quarter of the year, valued as one of 1, 2, 3, or 4.\n\n                    Also interesting for partitioning.\n\n                - relative_date_PCIAT is how many days since PCIAT was administered\n\n                    Feels like it wouldn't matter?\n\n        \"\"\"\n\n        return self._get_raw_features(subject_id)\n\n    \n\n    def engineer_features_for_subject(self, subject_id: str) -> dict[str, float]:\n\n        \"\"\"Calculate a dictionary of engineered features for a single subject.\n\n        \n\n        Arguments:\n\n            - subject_id: The ID of a subject who has parquet data (such as \"00115b9f\")\n\n            - plot (Optional): Plot features for this subject for helper functions which support this ability.\n\n            \n\n        Returns:\n\n            - A dictionary of relevant features from this parquet. Keys are feature names, values are feature values. \n\n        \"\"\"\n\n        return self._engineer_features_for_subject(subject_id)\n\n    \n\n    def engineer_features_for_subjects(self, subject_ids: Optional[list[str]] = None, parallel: bool = True) -> pd.DataFrame:\n\n        \"\"\"Calculate a pd.DataFrame of engineered features for ALL available subjects.\n\n        \n\n        Arguments:\n\n            - subject_ids (Optional) - \n\n                List of parquets to engineer features for. \n\n                Default behavior is to engineer features for all subjects.\n\n            - parallel (Optional) - \n\n                Set to True (default behavior) to process parquets in parallel.\n\n                Set to False to process parquets in series via a loop.\n\n    \n\n        Returns:\n\n            - A table whose columns are parquet-derived features, and whose rows are subjects indexed by their ID.\n\n        \"\"\"\n\n        return self._engineer_features_for_subjects(subject_ids, parallel)\n\n    \n\n    # --- END OF PUBLIC METHODS ---\n\n    \n\n    # --- PRIVATE METHODS --- (only need to interact with these if you want to modify how parquets are read)\n\n    \n\n    @property\n\n    def _parquet_ids(self) -> list[str]:\n\n        \"\"\"\n\n        A list of subject ID's which have parquet data available.\n\n        \n\n        See `parquet_ids` property for more information.\n\n        \"\"\"\n\n        return list(self._parquet_paths.keys())\n\n    \n\n    def _get_raw_features(self, subject_id: str) -> pd.DataFrame:\n\n        \"\"\"\n\n        Return a DataFrame of the given subject's raw parquet data.\n\n        \n\n        See `get_raw_features` method for more information.\n\n        \"\"\"\n\n        parquet_path = self._subject_id_to_path(subject_id)\n\n        raw_parquet = pd.read_parquet(parquet_path)\n\n        raw_parquet.set_index('step', inplace=True)\n\n        return raw_parquet\n\n        \n\n    def _engineer_features_for_subjects(self, subject_ids: Optional[list[str]] = None, parallel: bool = True) -> pd.DataFrame:\n\n        \"\"\"\n\n        Retrieve relevant features from each parquet, and make a DataFrame out of these features.\n\n\n\n        See `engineer_features_for_subjects` method for more information.\n\n        \"\"\"\n\n        \n\n        if subject_ids is None:\n\n            subject_ids = self._parquet_ids\n\n        \n\n        if parallel:\n\n            # Extract parquet data from their files in parallel.\n\n            with ThreadPoolExecutor() as executor:\n\n                all_subjects_parquet_features: list[dict[str,float]] = list(tqdm(\n\n                    executor.map(lambda subject_id: self._engineer_features_for_subject(subject_id), subject_ids), total=len(subject_ids), desc=\"Engineering features from parquets\"\n\n                ))\n\n        else:\n\n            # Extract parquet data from their files in series. This may be easier to interpret.\n\n            all_subjects_parquet_features: list[dict[str,float]] = []\n\n            for subject_id in tqdm(subject_ids, total=len(subject_ids), desc=\"Engineering features from parquets\"):\n\n                parquet_features = self._engineer_features_for_subject(subject_id)\n\n                all_subjects_parquet_features.append(parquet_features)\n\n                \n\n        # Convert the list of  to a DataFrame\n\n        pqdf = pd.DataFrame(all_subjects_parquet_features).set_index(\"id\")\n\n\n\n        return pqdf\n\n    \n\n    @staticmethod\n\n    def _path_to_subject_id(path: str) -> str:\n\n        \"\"\"\n\n        Extract a subject's ID from the path to their parquet data.\n\n        This subject ID is in the same format as the subject ID's in `train.csv` or `test.csv`.\n\n        For example, \"00115b9f\" is a valid subject ID that this function might return.\n\n        \"\"\"\n\n        return os.path.basename(os.path.dirname(path)).split(\"=\")[1]\n\n    \n\n    def _subject_id_to_path(self, subject_id: str) -> str:\n\n        \"\"\"\n\n        Get the path to the given subject's parquet data.\n\n        \"\"\"\n\n        return self._parquet_paths[subject_id]\n\n\n\n    @staticmethod\n\n    def _mask_times(times: pd.Series, start: str, end: Optional[str] = None) -> pd.Series:\n\n        \"\"\"\n\n        Convenience function for producing a mask on a Series of timestamps between the provided `start` and `end` times.\n\n        This is done in the range: [`start`, `end`). i.e. the `start` time is included, but the `end` time is not.\n\n\n\n        The `start` and `end` times should be defined in 24-hour time in this exact format: \"HH:MM:SS\"\n\n        For example, \"18:59:00\" and \"06:01:30\" are valid.\n\n        \n\n        Arguments:\n\n            - times - A Series of pandas datetimes.\n\n            - start - A start time whose format is specified above.\n\n            - end (Optional) - An end time whose format is specified above. \n\n                If `end` is omitted, then the mask is true for all entries after `start`.\n\n        \n\n        Returns:\n\n            - A Series of boolean values which is True-valued for times in [`start`, `end`).\n\n        \"\"\"\n\n        # Ensure the time format is right.\n\n        for cutoff in [start, end]:\n\n            assert bool(re.match(r'^(?:[01]\\d|2[0-3]):[0-5]\\d:[0-5]\\d$', cutoff)), Exception(f\"Provide `{cutoff}` in 24-hour HH:MM:SS format.\")\n\n\n\n        # Return the mask.\n\n        after_start = times >= pd.Timestamp(\"1970-01-01 \" + start)\n\n        if end is None:\n\n            return after_start\n\n        \n\n        after_end = times < pd.Timestamp(\"1970-01-01 \" + end)\n\n        return after_start & after_end\n\n    \n\n    # --- \n\n    # The remaining methods here are used to derive features from one subject's `parquet` DataFrame.\n\n    # ---\n\n    \n\n    @staticmethod\n\n    def _create_hourly_features(parquet: pd.DataFrame, signals: list[str]) -> dict[str, float]:\n\n        \"\"\"\n\n        Create features related to average signal values within each hour of the parquet data.\n\n        \n\n        Arguments:\n\n            - parquet, a subject's parquet data.\n\n                Precondition: The `hour` column has been defined.\n\n            - signals, a list of column names in `parquet`.\n\n            \n\n        Returns:\n\n            - A dictionary containing the following key-value pairs:\n\n                - For each of the 24 hours and for each signal in signals, the \n\n        \n\n        Modifications to `parquet`:\n\n            None.\n\n        \"\"\"\n\n        hourly_features = {}\n\n        \n\n        # Take the mean within each hour group.\n\n        hourly_average_df = parquet.groupby('hour')[signals].mean()\n\n        hours = hourly_average_df.index\n\n        \n\n        # Approximate derivative as first-order difference.\n\n        hourly_average_derivative_df = hourly_average_df.diff()\n\n        \n\n        for signal in signals:\n\n            # Add the hourly averages as features.\n\n            for hour in hours:\n\n                hourly_features[f'{signal}_hr{hour}_avg'] = hourly_average_df.at[hour, signal]\n\n                \n\n            # Record which hours the maximum & minimum average values occur at.\n\n            hourly_features[f'{signal}_min_hr'] = hourly_average_df[signal].idxmin()\n\n            hourly_features[f'{signal}_max_hr'] = hourly_average_df[signal].idxmax()\n\n            \n\n            # Record which hours the maximum & minimum changes in average values occur at.\n\n            hourly_features[f'{signal}_diffmin_hr'] = hourly_average_derivative_df[signal].idxmin()\n\n            hourly_features[f'{signal}_diffmax_hr'] = hourly_average_derivative_df[signal].idxmax()\n\n\n\n        return hourly_features\n\n    \n\n    def _engineer_features_for_subject(self, subject_id: str) -> dict[str, float]:\n\n        \"\"\"Calculate relevant features based on one subject's parquet file.\n\n\n\n        See `engineer_features_for_subject` method for more information.\n\n        \"\"\"\n\n\n\n        parquet: pd.DataFrame = self._get_raw_features(subject_id)\n\n        \n\n        # Prepare columns that will be used across multiple preprocessing\n\n        \n\n        # Interpret time of day as nanoseconds and cast to pandas datetime type.\n\n        # Ignore the 1970-01-01 component - this column only represents the time of day.\n\n        parquet['time_of_day'] = pd.to_datetime(parquet['time_of_day'], unit='ns')\n\n\n\n        # Create hour bin to group data by.\n\n        parquet['hour'] = parquet['time_of_day'].dt.hour\n\n        \n\n        # Create smoothed signals using a small window to reduce noise.\n\n        signals = ['enmo', 'anglez', 'light']\n\n        for signal in signals:\n\n            parquet[\"smooth_\"+signal] = parquet[signal].rolling(window=5).mean()\n\n        \n\n        # Create more features.\n\n        features = {\"id\": subject_id}\n\n        \n\n        # **********************************************\n\n        # ***  NEW PARQUET PROCESSING CODE GOES HERE ***\n\n        # **********************************************\n\n        \n\n        # Use parquet to bin a few signals on an hourly basis, take the average within each hour, and derive some features from that.\n\n        hourly_averages = self._create_hourly_features(parquet, ['smooth_enmo', 'smooth_anglez', 'smooth_light'])\n\n        features.update(hourly_averages)\n\n        \n\n        # (more features to be calculated...)\n\n        \n\n        # ***********************************************\n\n        # ******* END NEW PARQUET PROCESSING CODE *******\n\n        # ***********************************************\n\n        return features","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nimport pandas as pd\n\nimport os\n\nfrom sklearn.metrics import cohen_kappa_score\n\n\n\n# As a simple approach:\n\n# - Only use labeled tabular data for now.\n\n# - Ignore parquet data.\n\n# - Ignore PCIAT columns which don't exist in test set (these are directly used to calculate sii, so we could consider them as intermediate targets to predict)\n\n# - One-hot encode strings.\n\n# - Impute missing numbers as mean of that feature. This includes string one-hot encodings for now.\n\n# - Avoid further prep by using XGBoost as model\n\n# - Use simple set-aside test set for local evaluation. Do not tune hyperparameters yet.\n\n\n\ndf_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n\n\n# NEW: Add parquet features.\n\npa_train = ParquetAnalyzer(\"train\")\n\npq_df_train = pa_train.engineer_features_for_subjects(parallel=True)\n\ndf_train = pd.concat(\n\n    [df_train, pq_df_train], \n\n    axis=1, join='outer')\n\n\n\npa_test = ParquetAnalyzer(\"test\")\n\npq_df_test = pa_test.engineer_features_for_subjects(parallel=True)\n\ndf_test = pd.concat(\n\n    [df_test, pq_df_test], \n\n    axis=1, join='outer')\n\n# END NEW\n\n\n\nstudies_to_drop = ['PCIAT', 'BIA', 'Season']\n\nfor study in studies_to_drop:\n\n    df_train = df_train.drop(columns=df_train.filter(like=study).columns)\n\n    df_test = df_test.drop(columns=df_test.filter(like=study).columns)\n\n\n\ndf_train = df_train.drop(columns=['id'], axis=1)\n\ndf_train_unlabeled = df_train.query('sii!=sii') # TODO - determine what to do with this. Clustering?\n\ndf_train_labeled = df_train.query('sii==sii')\n\nX_train_labeled, Y_train_labeled = df_train_labeled.drop('sii', axis=1), df_train_labeled['sii']\n\n\n\ndf_test, index_test = df_test.drop(columns=['id'], axis=1), df_test['id']","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:27.872246Z","iopub.status.busy":"2024-10-09T15:27:27.871633Z","iopub.status.idle":"2024-10-09T15:27:28.002319Z","shell.execute_reply":"2024-10-09T15:27:28.000272Z","shell.execute_reply.started":"2024-10-09T15:27:27.87218Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\n\n\n\ndef apply_xgboost_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    model = xgb.XGBClassifier(tree_method=\"hist\")\n\n    model.fit(trainX, trainY)\n\n    Y_pred = model.predict(testX)\n\n    \n\n    return Y_pred, model","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.006648Z","iopub.status.busy":"2024-10-09T15:27:28.006155Z","iopub.status.idle":"2024-10-09T15:27:28.015449Z","shell.execute_reply":"2024-10-09T15:27:28.013711Z","shell.execute_reply.started":"2024-10-09T15:27:28.0066Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\n\n\ndef apply_random_forest_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    clf = RandomForestClassifier(random_state=0)\n\n    clf.fit(trainX, trainY)\n\n    Y_pred = clf.predict(testX)\n\n    \n\n    return Y_pred, clf","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.018683Z","iopub.status.busy":"2024-10-09T15:27:28.018131Z","iopub.status.idle":"2024-10-09T15:27:28.033557Z","shell.execute_reply":"2024-10-09T15:27:28.031895Z","shell.execute_reply.started":"2024-10-09T15:27:28.018623Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.discriminant_analysis import LinearDiscriminantAnalysis\n\n\n\ndef apply_lda_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    clf = LinearDiscriminantAnalysis()\n\n    clf.fit(trainX, trainY)\n\n    Y_pred = clf.predict(testX)\n\n    \n\n    return Y_pred, clf","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.038433Z","iopub.status.busy":"2024-10-09T15:27:28.037489Z","iopub.status.idle":"2024-10-09T15:27:28.046902Z","shell.execute_reply":"2024-10-09T15:27:28.045809Z","shell.execute_reply.started":"2024-10-09T15:27:28.038371Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis\n\n\n\ndef apply_qda_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    clf = QuadraticDiscriminantAnalysis()\n\n    clf.fit(trainX, trainY)\n\n    Y_pred = clf.predict(testX)\n\n    \n\n    return Y_pred, clf","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.049632Z","iopub.status.busy":"2024-10-09T15:27:28.049027Z","iopub.status.idle":"2024-10-09T15:27:28.063331Z","shell.execute_reply":"2024-10-09T15:27:28.061475Z","shell.execute_reply.started":"2024-10-09T15:27:28.049577Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\n\n\ndef apply_gaussian_nb_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    clf = GaussianNB()\n\n    clf.fit(trainX, trainY)\n\n    Y_pred = clf.predict(testX)\n\n    \n\n    return Y_pred, clf","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.065906Z","iopub.status.busy":"2024-10-09T15:27:28.0651Z","iopub.status.idle":"2024-10-09T15:27:28.076276Z","shell.execute_reply":"2024-10-09T15:27:28.075097Z","shell.execute_reply.started":"2024-10-09T15:27:28.065841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.svm import SVC\n\n\n\ndef apply_rbf_svm_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    clf = SVC()\n\n    clf.fit(trainX, trainY)\n\n    Y_pred = clf.predict(testX)\n\n    \n\n    return Y_pred, clf","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.079197Z","iopub.status.busy":"2024-10-09T15:27:28.078107Z","iopub.status.idle":"2024-10-09T15:27:28.086961Z","shell.execute_reply":"2024-10-09T15:27:28.085808Z","shell.execute_reply.started":"2024-10-09T15:27:28.07914Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier\n\n\n\ndef apply_adaboost_model(tX, tY, vX):\n\n    trainX = tX.copy()\n\n    trainY = tY.copy()\n\n    testX = vX.copy()\n\n    \n\n    clf = AdaBoostClassifier()\n\n    clf.fit(trainX, trainY)\n\n    Y_pred = clf.predict(testX)\n\n    \n\n    return Y_pred, clf","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.088985Z","iopub.status.busy":"2024-10-09T15:27:28.088514Z","iopub.status.idle":"2024-10-09T15:27:28.099975Z","shell.execute_reply":"2024-10-09T15:27:28.098511Z","shell.execute_reply.started":"2024-10-09T15:27:28.08892Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\n\nimport keras\n\n\n\ndef apply_stack_nn_model(trainX, trainY, testX, learning_rate=0.007, n_epochs=5, verbose=False):\n\n    # create an 80/20 test split to train the MLP\n\n    tX, vX, tY, vY = train_test_split(trainX, trainY, test_size=0.2, random_state=0)\n\n    \n\n    # impute missing values using means\n\n    numeric_imputer = SimpleImputer(missing_values=np.nan, strategy='mean', copy=True)\n\n    tX = numeric_imputer.fit_transform(tX)\n\n    vX = numeric_imputer.transform(vX)\n\n    testX = numeric_imputer.transform(testX)\n\n    \n\n    Init = keras.initializers.RandomNormal(seed=0)\n\n\n\n    # creates an MLP with 2 20-neuron hidden layers and an output to calculate between 0-1, which then gets scaled up to 0-5\n\n    m = keras.models.Sequential()\n\n    m.add(keras.layers.Flatten(input_shape=[tX.shape[1]], name=\"input\"))      # creates input layer with n_features perceptrons\n\n    m.add(keras.layers.Dense(20, activation=\"relu\", kernel_initializer=Init, name='hidden_1'))    # creates hidden layer with 10 perceptrons\n\n    m.add(keras.layers.Dense(20, activation=\"relu\", kernel_initializer=Init, name='hidden_2'))    # creates hidden layer with 10 perceptrons\n\n    m.add(keras.layers.Dense(1, activation='sigmoid', kernel_initializer=Init, name='output'))  # create output layer with 1 perceptron (regressor with bins)\n\n    if verbose: m.summary()   # displays all model alyers\n\n    m.compile(loss='binary_crossentropy', optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate))\n\n\n\n    if verbose:\n\n        hist = m.fit(tX, tY/5, epochs=n_epochs, validation_data=(vX, vY/5))\n\n    else:\n\n        hist = m.fit(tX, tY/5, epochs=n_epochs, validation_data=(vX, vY/5), verbose=0)\n\n\n\n    pred_out = m.predict(testX) * 5\n\n    Y_pred = pred_out.copy()\n\n    Y_pred = np.round(Y_pred)\n\n\n\n    return Y_pred, hist","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.102431Z","iopub.status.busy":"2024-10-09T15:27:28.101908Z","iopub.status.idle":"2024-10-09T15:27:28.138039Z","shell.execute_reply":"2024-10-09T15:27:28.136825Z","shell.execute_reply.started":"2024-10-09T15:27:28.10237Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\n\nimport keras\n\n\n\ndef apply_triang_nn_model(trainX, trainY, testX, learning_rate=0.007, n_epochs=5, verbose=False):\n\n    # create an 80/20 test split to train the MLP\n\n    tX, vX, tY, vY = train_test_split(trainX, trainY, test_size=0.2, random_state=0)\n\n    \n\n    Init = keras.initializers.RandomNormal(seed=0)\n\n\n\n    # creates an MLP with 3 hidden layers and an output to calculate between 0-1, which then gets scaled up to 0-5\n\n    m = keras.models.Sequential()\n\n    m.add(keras.layers.Flatten(input_shape=[tX.shape[1]], name=\"input\"))      # creates input layer with n_features perceptrons\n\n    m.add(keras.layers.Dense(16, activation=\"relu\", kernel_initializer=Init, name='hidden_1'))    # creates hidden layer with 16 perceptrons\n\n    m.add(keras.layers.Dense(8, activation=\"relu\", kernel_initializer=Init, name='hidden_2'))    # creates hidden layer with 18 perceptrons\n\n    m.add(keras.layers.Dense(4, activation=\"relu\", kernel_initializer=Init, name='hidden_3'))    # creates hidden layer with 4 perceptrons\n\n    m.add(keras.layers.Dense(1, activation='sigmoid', kernel_initializer=Init, name='output'))  # create output layer with 1 perceptron (regressor with bins)\n\n    if verbose: m.summary()   # displays all model alyers\n\n    m.compile(loss='binary_crossentropy', optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate))\n\n\n\n    if verbose:\n\n        hist = m.fit(tX, tY/5, epochs=n_epochs, validation_data=(vX, vY/5))\n\n    else:\n\n        hist = m.fit(tX, tY/5, epochs=n_epochs, validation_data=(vX, vY/5), verbose=0)\n\n\n\n    pred_out = m.predict(testX) * 5\n\n    Y_pred = pred_out.copy()\n\n    Y_pred = np.round(Y_pred)\n\n\n\n    return Y_pred, hist","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.142013Z","iopub.status.busy":"2024-10-09T15:27:28.141617Z","iopub.status.idle":"2024-10-09T15:27:28.158427Z","shell.execute_reply":"2024-10-09T15:27:28.157014Z","shell.execute_reply.started":"2024-10-09T15:27:28.141966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\nfrom sklearn.decomposition import PCA\n\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.impute import SimpleImputer\n\nimport matplotlib.pyplot as plt\n\nimport time\n\n\n\nif run_cv:\n\n    tf.random.set_seed(0)\n\n    np.random.seed(0)\n\n    tf.keras.utils.set_random_seed(0)\n\n    \n\n    # model parameters\n\n    n_epochs = 50\n\n    n_samples = 'all'\n\n    learning_rate = 0.0007\n\n    verbose = False\n\n    n_components = 26         # number of pca components to keep, this was found through testing\n\n    \n\n    # parameters for K-fold cross-validation\n\n    k = 5\n\n    kf = KFold(n_splits=k, shuffle=True, random_state=1)\n\n    \n\n    if n_samples == 'all':\n\n        X_train_labeled, Y_train_labeled = df_train_labeled.drop('sii', axis=1), df_train_labeled['sii']\n\n    else:\n\n        tXY = df_train_labeled.sample(n=n_samples, random_state=0)\n\n        X_train_labeled, Y_train_labeled = tXY.drop('sii', axis=1), tXY['sii']\n\n    \n\n    i = 0\n\n    average_score = 0\n\n    average_runtime = 0\n\n\n\n    # start k-fold CV\n\n    for train, test in kf.split(X_train_labeled, Y_train_labeled):\n\n        i += 1\n\n        if verbose: print(\"Starting fold %s/%s -----------------------------------\" % (i, k))\n\n        start_time = time.time()\n\n\n\n        # gets train and test sets\n\n        X_train = X_train_labeled.iloc[train].copy()\n\n        Y_train = Y_train_labeled.iloc[train].copy()\n\n\n\n        X_test = X_train_labeled.iloc[test].copy()\n\n        Y_test = Y_train_labeled.iloc[test].copy()\n\n\n\n        numeric_imputer = SimpleImputer(missing_values=np.nan, strategy='mean', copy=True)\n\n        X_train = numeric_imputer.fit_transform(X_train)\n\n        X_test = numeric_imputer.transform(X_test)\n\n\n\n        scaler = StandardScaler()\n\n        X_train = pd.DataFrame(scaler.fit_transform(X_train))\n\n        X_test = pd.DataFrame(scaler.transform(X_test))\n\n\n\n        pca = PCA(n_components=n_components)\n\n        X_train = pd.DataFrame(pca.fit_transform(X_train))\n\n        X_test = pd.DataFrame(pca.transform(X_test))\n\n\n\n        # trains the model on X_train, Y_train, applies it to X_test\n\n        Y_pred, hist = apply_triang_nn_model(X_train, Y_train, X_test, learning_rate=learning_rate, n_epochs=n_epochs, verbose = verbose)\n\n\n\n        # scores the model and times it\n\n        score = cohen_kappa_score(Y_test, Y_pred, weights='quadratic')\n\n        average_score += score\n\n        if verbose: print(\"Cohen Kappa Score: %.6f\" % score)\n\n\n\n        runtime = time.time() - start_time\n\n        average_runtime += runtime\n\n        if verbose: print(\"Runtime: %.6f s\" % runtime)\n\n\n\n    # outputs average results across all K folds\n\n    average_score = average_score / k\n\n    average_runtime = average_runtime / k\n\n    print(\"Final Results -----------------------------\")\n\n    print(\"Average Score: %.6f\" % average_score)\n\n    print(\"Average Runtime: %.6f s\" % average_runtime)","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.160542Z","iopub.status.busy":"2024-10-09T15:27:28.160103Z","iopub.status.idle":"2024-10-09T15:27:28.183407Z","shell.execute_reply":"2024-10-09T15:27:28.182137Z","shell.execute_reply.started":"2024-10-09T15:27:28.160497Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if run_output:\n\n    start_time = time.time()\n\n    \n\n    tf.random.set_seed(0)\n\n    np.random.seed(0)\n\n    tf.keras.utils.set_random_seed(0)\n\n    \n\n    # model parameters\n\n    n_epochs = 16             # noticed val_loss increases after this. need to implement early sstopping\n\n    n_samples = 'all'\n\n    learning_rate = 0.0007\n\n    verbose = True\n\n    n_components = 26         # number of pca components to keep, this was found through testing\n\n    \n\n    if n_samples == 'all':\n\n        X_train, Y_train = df_train_labeled.drop('sii', axis=1), df_train_labeled['sii']\n\n    else:\n\n        tXY = df_train_labeled.sample(n=n_samples, random_state=0)\n\n        X_train, Y_train = tXY.drop('sii', axis=1), tXY['sii']\n\n        \n\n    X_test = df_test.copy()\n\n    \n\n    numeric_imputer = SimpleImputer(missing_values=np.nan, strategy='mean', copy=True)\n\n    X_train = numeric_imputer.fit_transform(X_train)\n\n    X_test = numeric_imputer.transform(X_test)\n\n\n\n    scaler = StandardScaler()\n\n    X_train = pd.DataFrame(scaler.fit_transform(X_train))\n\n    X_test = pd.DataFrame(scaler.transform(X_test))\n\n\n\n    pca = PCA(n_components=n_components)\n\n    X_train = pd.DataFrame(pca.fit_transform(X_train))\n\n    X_test = pd.DataFrame(pca.transform(X_test))\n\n    \n\n    # applies the model to the full test set and outputs it to \"submission.csv\"\n\n    Y_pred, hist = apply_triang_nn_model(X_train, Y_train, X_test, learning_rate=learning_rate, n_epochs=n_epochs, verbose = verbose)\n\n    #testY_pred = apply_model(X_train_labeled, Y_train_labeled, df_test)\n\n    pY = pd.DataFrame(Y_pred, index=index_test, columns=['sii'])\n\n    pY.astype(int).to_csv('submission.csv', index_label='id')\n\n    print(\"Runtime: %.6f s\" % (time.time() - start_time))","metadata":{"execution":{"iopub.execute_input":"2024-10-09T15:27:28.185163Z","iopub.status.busy":"2024-10-09T15:27:28.184744Z","iopub.status.idle":"2024-10-09T15:27:33.056018Z","shell.execute_reply":"2024-10-09T15:27:33.054873Z","shell.execute_reply.started":"2024-10-09T15:27:28.185121Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Submission 1 (Justin's Code)**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n\nimport numpy as np\n\nimport pandas as pd\n\nimport os\n\nfrom sklearn.impute import SimpleImputer\n\nfrom sklearn.metrics import cohen_kappa_score\n\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport xgboost as xgb\n\n\n\n# As a simple approach:\n\n# - Only use labeled tabular data for now.\n\n# - Ignore parquet data.\n\n# - Ignore PCIAT columns which don't exist in test set (these are directly used to calculate sii, so we could consider them as intermediate targets to predict)\n\n# - One-hot encode strings.\n\n# - Impute missing numbers as mean of that feature. This includes string one-hot encodings for now.\n\n# - Avoid further prep by using XGBoost as model\n\n# - Use simple set-aside test set for local evaluation. Do not tune hyperparameters yet.\n\n\n\ndf_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n\n\ndf_train = df_train.drop(columns=df_train.filter(like='PCIAT').columns)\n\ndf_train = df_train.drop(columns=['id'], axis=1)\n\ndf_train_unlabeled = df_train.query('sii!=sii') # TODO - determine what to do with this. Clustering?\n\ndf_train_labeled = df_train.query('sii==sii')\n\nX_train_labeled, Y_train_labeled = df_train_labeled.drop('sii', axis=1), df_train_labeled['sii']\n\n\n\ndf_test, index_test = df_test.drop(columns=['id'], axis=1), df_test['id']\n\n\n\n# Do not use unlabeled data for now.\n\ntX, vX, tY, vY = train_test_split(X_train_labeled, Y_train_labeled, test_size=0.2, random_state=42)\n\n\n\nstring_encoder = OneHotEncoder(drop=None, handle_unknown='ignore')\n\ntX = string_encoder.fit_transform(tX)\n\nvX = string_encoder.transform(vX)\n\ndf_test = string_encoder.transform(df_test)\n\n\n\nnumeric_imputer = SimpleImputer(missing_values=np.nan, strategy='mean', copy=True)\n\ntX = numeric_imputer.fit_transform(tX)\n\nvX = numeric_imputer.transform(vX)\n\ndf_test = numeric_imputer.transform(df_test)\n\n\n\nmodel = xgb.XGBClassifier(tree_method=\"hist\")\n\nmodel.fit(tX, tY)\n\nvY_pred = model.predict(vX)\n\ntestY_pred = model.predict(df_test)\n\n\n\nprint(cohen_kappa_score(vY, vY_pred, weights='quadratic'))\n\n\n\npY = pd.DataFrame(testY_pred, index=index_test, columns=['sii'])\n\npY.astype(int).to_csv('submission.csv', index_label='id')\n\n\"\"\"","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2024-10-09T15:27:33.058349Z","iopub.status.busy":"2024-10-09T15:27:33.057932Z","iopub.status.idle":"2024-10-09T15:27:33.069424Z","shell.execute_reply":"2024-10-09T15:27:33.068197Z","shell.execute_reply.started":"2024-10-09T15:27:33.05828Z"},"trusted":true},"outputs":[],"execution_count":null}]}