{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:43:17.216383Z","iopub.execute_input":"2024-10-11T09:43:17.217197Z","iopub.status.idle":"2024-10-11T09:43:18.332822Z","shell.execute_reply.started":"2024-10-11T09:43:17.217159Z","shell.execute_reply":"2024-10-11T09:43:18.331839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport pyarrow.parquet as pq\nimport numpy as np\nimport os\n\n# Load tabular data\ntrain_tabular = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_tabular = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n\n# Load time-series (accelerometer) data\n# Assuming parquet files are partitioned by 'id', we iterate through available ids.\ndef load_parquet_data(parquet_folder, ids):\n    data = {}\n    for id_ in ids:\n        file_path = os.path.join(parquet_folder, f\"id={id_}.parquet\")\n        if os.path.exists(file_path):\n            data[id_] = pq.read_table(file_path).to_pandas()\n    return data\n\n# Assuming the ids in train_tabular for participants with accelerometer data\ntrain_ids = train_tabular['id'].unique()\nparquet_folder = \"series_train.parquet\"\ntrain_time_series = load_parquet_data(parquet_folder, train_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:44:52.097945Z","iopub.execute_input":"2024-10-11T09:44:52.098356Z","iopub.status.idle":"2024-10-11T09:44:52.186567Z","shell.execute_reply.started":"2024-10-11T09:44:52.098320Z","shell.execute_reply":"2024-10-11T09:44:52.185576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:44:53.833628Z","iopub.execute_input":"2024-10-11T09:44:53.834470Z","iopub.status.idle":"2024-10-11T09:44:53.838631Z","shell.execute_reply.started":"2024-10-11T09:44:53.834432Z","shell.execute_reply":"2024-10-11T09:44:53.837555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Function to fill missing data\ndef fill_missing_data(df):\n    for col in df.columns:\n        if df[col].dtype in ['float64', 'int64']:\n            # Fill numerical data with the mean\n            df[col].fillna(df[col].mean(), inplace=True)\n        elif col=='id':\n            continue\n\n        else:\n            # Convert categorical data to numeric using a mapping\n            unique_values = df[col].unique()\n            # Create a mapping of unique values to numeric\n            mapping = {value: index for index, value in enumerate(unique_values)}\n            # Map the categorical values to numbers\n            df[col] = df[col].map(mapping)\n            # Fill NaN values with the rounded mean of the numeric values\n            df[col].fillna(round(df[col].mean()), inplace=True)\n    return df\n\n# Fill missing values in both train and test datasets\ntrain_tabular = fill_missing_data(train_tabular)\ntest_tabular = fill_missing_data(test_tabular)\n\n# Check the result\nprint(train_tabular.head())\nprint(test_tabular.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:46:12.738581Z","iopub.execute_input":"2024-10-11T09:46:12.738963Z","iopub.status.idle":"2024-10-11T09:46:12.828462Z","shell.execute_reply.started":"2024-10-11T09:46:12.738929Z","shell.execute_reply":"2024-10-11T09:46:12.827489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tabular.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:46:17.481521Z","iopub.execute_input":"2024-10-11T09:46:17.482235Z","iopub.status.idle":"2024-10-11T09:46:17.488960Z","shell.execute_reply.started":"2024-10-11T09:46:17.482194Z","shell.execute_reply":"2024-10-11T09:46:17.487898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:46:28.460923Z","iopub.execute_input":"2024-10-11T09:46:28.461803Z","iopub.status.idle":"2024-10-11T09:46:28.466560Z","shell.execute_reply.started":"2024-10-11T09:46:28.461763Z","shell.execute_reply":"2024-10-11T09:46:28.465658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_feature = train_tabular.select_dtypes(include=['number']).columns.to_numpy()\nprint(num_feature)\nPCIAT = ['PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04',\n         'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', \n         'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n         'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16',\n         'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', \n         'PCIAT-PCIAT_Total', 'PCIAT-Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:46:31.286103Z","iopub.execute_input":"2024-10-11T09:46:31.286504Z","iopub.status.idle":"2024-10-11T09:46:31.296771Z","shell.execute_reply.started":"2024-10-11T09:46:31.286469Z","shell.execute_reply":"2024-10-11T09:46:31.295829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tabular.drop(PCIAT, axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:46:33.668747Z","iopub.execute_input":"2024-10-11T09:46:33.669405Z","iopub.status.idle":"2024-10-11T09:46:33.675975Z","shell.execute_reply.started":"2024-10-11T09:46:33.669366Z","shell.execute_reply":"2024-10-11T09:46:33.675005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install lightgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:46:40.572344Z","iopub.execute_input":"2024-10-11T09:46:40.573013Z","iopub.status.idle":"2024-10-11T09:47:11.830314Z","shell.execute_reply.started":"2024-10-11T09:46:40.572975Z","shell.execute_reply":"2024-10-11T09:47:11.829015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom lightgbm import LGBMRegressor  # Import LightGBM regressor\nfrom sklearn.model_selection import train_test_split\n\n# Assuming you already have your dataframe loaded as train_tabular\n\ntarget = 'sii'\niddtest = test_tabular['id']\niddtrain = train_tabular['id']\nfeatures = train_tabular.columns.drop([target, 'id'])  # Drop the target and 'id' for modeling\n\n# Separate features and target variable\nX = train_tabular[features]\ny = train_tabular[target]\n\n# Handle missing values for numerical and categorical columns\nnum_cols = X.select_dtypes(include=['float64', 'int64']).columns\ncat_cols = X.select_dtypes(include=['object']).columns\n\n# Pipeline for numerical features: imputing missing values and scaling\nnum_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())\n])\n\n# Pipeline for categorical features: imputing missing values and encoding\ncat_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Combine both pipelines into a ColumnTransformer\npreprocessor = ColumnTransformer(transformers=[\n    ('num', num_pipeline, num_cols),\n    ('cat', cat_pipeline, cat_cols)\n])\n\n# Model pipeline with preprocessing and LGBMRegressor\nmodel_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('regressor', LGBMRegressor(n_estimators=500, random_state=11))  # Use LightGBM regressor here\n])\n\n# Train-test split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=33)\n\n# Fit the model\nmodel_pipeline.fit(X_train, y_train)\n\n# Evaluate the model\ntrain_score = model_pipeline.score(X_train, y_train)\nval_score = model_pipeline.score(X_val, y_val)\n\nprint(f\"Training Score: {train_score}\")\nprint(f\"Validation Score: {val_score}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:47:15.498277Z","iopub.execute_input":"2024-10-11T09:47:15.498695Z","iopub.status.idle":"2024-10-11T09:47:16.607236Z","shell.execute_reply.started":"2024-10-11T09:47:15.498657Z","shell.execute_reply":"2024-10-11T09:47:16.606152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = test_tabular[features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:47:18.382275Z","iopub.execute_input":"2024-10-11T09:47:18.382675Z","iopub.status.idle":"2024-10-11T09:47:18.388438Z","shell.execute_reply.started":"2024-10-11T09:47:18.382636Z","shell.execute_reply":"2024-10-11T09:47:18.387499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = model_pipeline.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:47:19.510417Z","iopub.execute_input":"2024-10-11T09:47:19.511170Z","iopub.status.idle":"2024-10-11T09:47:19.522313Z","shell.execute_reply.started":"2024-10-11T09:47:19.511133Z","shell.execute_reply":"2024-10-11T09:47:19.521438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': iddtest,\n    'sii': abs(test_predictions.round())\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:47:20.682149Z","iopub.execute_input":"2024-10-11T09:47:20.682534Z","iopub.status.idle":"2024-10-11T09:47:20.687575Z","shell.execute_reply.started":"2024-10-11T09:47:20.682497Z","shell.execute_reply":"2024-10-11T09:47:20.686640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:47:21.778834Z","iopub.execute_input":"2024-10-11T09:47:21.779718Z","iopub.status.idle":"2024-10-11T09:47:21.792442Z","shell.execute_reply.started":"2024-10-11T09:47:21.779680Z","shell.execute_reply":"2024-10-11T09:47:21.791427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T09:47:26.090458Z","iopub.execute_input":"2024-10-11T09:47:26.090846Z","iopub.status.idle":"2024-10-11T09:47:26.097196Z","shell.execute_reply.started":"2024-10-11T09:47:26.090810Z","shell.execute_reply":"2024-10-11T09:47:26.096359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}