{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":212457019,"sourceType":"kernelVersion"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\n\ntrain_data = pd.read_csv('/kaggle/input/handling-sii/impute_train_data.csv', index_col='id')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv', index_col='id')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:37:23.251996Z","iopub.execute_input":"2024-12-18T06:37:23.252593Z","iopub.status.idle":"2024-12-18T06:37:24.691597Z","shell.execute_reply.started":"2024-12-18T06:37:23.252540Z","shell.execute_reply":"2024-12-18T06:37:24.690305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = train_data.copy()\ntest_df = test_data.copy()\n\ntrain_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:37:24.693896Z","iopub.execute_input":"2024-12-18T06:37:24.694330Z","iopub.status.idle":"2024-12-18T06:37:24.706669Z","shell.execute_reply.started":"2024-12-18T06:37:24.694293Z","shell.execute_reply":"2024-12-18T06:37:24.705236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = train_data.columns.tolist()\ntest_cols = test_data.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:37:24.709251Z","iopub.execute_input":"2024-12-18T06:37:24.709654Z","iopub.status.idle":"2024-12-18T06:37:24.726869Z","shell.execute_reply.started":"2024-12-18T06:37:24.709619Z","shell.execute_reply":"2024-12-18T06:37:24.725277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = test_cols.copy()\n\nnum_features = [f for f in features if test_df[f].dtype == 'float' or f == 'Basic_Demos-Age']\ncat_features = [f for f in features if f not in num_features]\n\nlen(features), len(num_features), len(cat_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:37:24.728894Z","iopub.execute_input":"2024-12-18T06:37:24.729393Z","iopub.status.idle":"2024-12-18T06:37:24.747865Z","shell.execute_reply.started":"2024-12-18T06:37:24.729346Z","shell.execute_reply":"2024-12-18T06:37:24.745925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:37:24.751895Z","iopub.execute_input":"2024-12-18T06:37:24.752437Z","iopub.status.idle":"2024-12-18T06:37:24.767421Z","shell.execute_reply.started":"2024-12-18T06:37:24.752379Z","shell.execute_reply":"2024-12-18T06:37:24.766159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    \n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data\n\ntrain_ts = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\ntest_ts = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove('id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:37:24.768639Z","iopub.execute_input":"2024-12-18T06:37:24.768983Z","iopub.status.idle":"2024-12-18T06:39:04.097091Z","shell.execute_reply.started":"2024-12-18T06:37:24.768942Z","shell.execute_reply":"2024-12-18T06:39:04.095984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import KNNImputer, SimpleImputer\nfrom sklearn.preprocessing import MinMaxScaler, OneHotEncoder\nfrom sklearn.feature_selection import SelectKBest, r_regression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:04.098791Z","iopub.execute_input":"2024-12-18T06:39:04.099236Z","iopub.status.idle":"2024-12-18T06:39:04.968569Z","shell.execute_reply.started":"2024-12-18T06:39:04.099171Z","shell.execute_reply":"2024-12-18T06:39:04.967296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_transformer = Pipeline(steps=[\n    ('KNNimputer', KNNImputer(n_neighbors=2, weights='uniform')),\n    ('MinMaxScaler', MinMaxScaler())\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:04.969988Z","iopub.execute_input":"2024-12-18T06:39:04.970489Z","iopub.status.idle":"2024-12-18T06:39:04.976172Z","shell.execute_reply.started":"2024-12-18T06:39:04.970456Z","shell.execute_reply":"2024-12-18T06:39:04.974769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='unknown')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:04.977640Z","iopub.execute_input":"2024-12-18T06:39:04.977991Z","iopub.status.idle":"2024-12-18T06:39:04.993161Z","shell.execute_reply.started":"2024-12-18T06:39:04.977958Z","shell.execute_reply":"2024-12-18T06:39:04.991926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ts_transformer = Pipeline(steps=[\n    ('MinMaxScaler', MinMaxScaler()),\n    ('imputer', SimpleImputer(strategy='median'))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:04.995168Z","iopub.execute_input":"2024-12-18T06:39:04.995631Z","iopub.status.idle":"2024-12-18T06:39:05.006043Z","shell.execute_reply.started":"2024-12-18T06:39:04.995583Z","shell.execute_reply":"2024-12-18T06:39:05.004734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(transformers=[\n    ('numerical', num_transformer, num_features),\n    ('categorical', cat_transformer, cat_features),\n    ('time_series', ts_transformer, time_series_cols)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:05.007490Z","iopub.execute_input":"2024-12-18T06:39:05.007854Z","iopub.status.idle":"2024-12-18T06:39:05.017425Z","shell.execute_reply.started":"2024-12-18T06:39:05.007822Z","shell.execute_reply":"2024-12-18T06:39:05.016186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\n\nparams1 = {  \n    \n    'metric'              :'rmse',\n    'objective'           :'regression',\n    'learning_rate'       : 0.04,\n    'max_depth'           : 12,\n    'num_leaves'          : 59,\n    'subsample'           : 0.70,\n    'colsample_bytree'    : 0.50,\n    'min_child_weight'    : 12, \n    'min_child_samples'   : 14,    \n    'reg_alpha'           : 0.23,\n    'reg_lambda'          : 0.36,\n}\nparams2 = {  \n    \n    'metric'              :'rmse',\n    'objective'           :'regression',\n    'learning_rate'       : 0.05,\n    'max_depth'           : 9,\n    'num_leaves'          : 59,\n    'subsample'           : 0.80,\n    'colsample_bytree'    : 0.50,\n    'min_child_weight'    : 12, \n    'min_child_samples'   : 14,  \n    'reg_alpha'           : 0.23,\n    'reg_lambda'          : 0.36,\n}\nparams3 = {  \n    \n    'metric'              :'rmse',\n    'objective'           :'regression',\n    'learning_rate'       : 0.046,\n    'max_depth'           : 12,\n    'num_leaves'          : 478,\n    'min_data_in_leaf'    : 13,\n    'feature_fraction'    : 0.893,\n    'bagging_fraction'    : 0.784,\n    'bagging_freq'        : 4,\n    'lambda_l1'           : 10, \n    'lambda_l2'           : 0.01, \n}\n\nmodel1 = LGBMRegressor(**params1, n_estimators=350, verbose=-1)\nmodel2 = LGBMRegressor(**params2, n_estimators=350, verbose=-1)\nmodel3 = LGBMRegressor(**params3, n_estimators=350, verbose=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:05.019402Z","iopub.execute_input":"2024-12-18T06:39:05.019846Z","iopub.status.idle":"2024-12-18T06:39:06.075950Z","shell.execute_reply.started":"2024-12-18T06:39:05.019799Z","shell.execute_reply":"2024-12-18T06:39:06.074739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline1 = Pipeline(steps=[\n    ('preprocess', preprocessor),\n    ('feature_selection', SelectKBest(score_func=r_regression, k=117)),\n    ('model', model1)\n])\npipeline2 = Pipeline(steps=[\n    ('preprocess', preprocessor),\n    ('feature_selection', SelectKBest(score_func=r_regression, k=117)),\n    ('model', model2)\n])\npipeline3 = Pipeline(steps=[\n    ('preprocess', preprocessor),\n    ('feature_selection', SelectKBest(score_func=r_regression, k=117)),\n    ('model', model3)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:06.077432Z","iopub.execute_input":"2024-12-18T06:39:06.078099Z","iopub.status.idle":"2024-12-18T06:39:06.085033Z","shell.execute_reply.started":"2024-12-18T06:39:06.078050Z","shell.execute_reply":"2024-12-18T06:39:06.083651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"main_train_df = pd.merge(train_df[features], train_ts, how=\"left\", on='id')\nmain_test_df = pd.merge(test_df, test_ts, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:06.090346Z","iopub.execute_input":"2024-12-18T06:39:06.090836Z","iopub.status.idle":"2024-12-18T06:39:06.138752Z","shell.execute_reply.started":"2024-12-18T06:39:06.090797Z","shell.execute_reply":"2024-12-18T06:39:06.137500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = main_train_df.copy()\ny = train_df['sii']\nXX = main_test_df.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:06.140658Z","iopub.execute_input":"2024-12-18T06:39:06.141058Z","iopub.status.idle":"2024-12-18T06:39:06.156474Z","shell.execute_reply.started":"2024-12-18T06:39:06.141026Z","shell.execute_reply":"2024-12-18T06:39:06.155181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline1.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:06.158059Z","iopub.execute_input":"2024-12-18T06:39:06.158530Z","iopub.status.idle":"2024-12-18T06:39:16.478253Z","shell.execute_reply.started":"2024-12-18T06:39:06.158477Z","shell.execute_reply":"2024-12-18T06:39:16.476960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline2.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:16.479680Z","iopub.execute_input":"2024-12-18T06:39:16.480511Z","iopub.status.idle":"2024-12-18T06:39:27.770716Z","shell.execute_reply.started":"2024-12-18T06:39:16.480478Z","shell.execute_reply":"2024-12-18T06:39:27.769382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline3.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:27.773080Z","iopub.execute_input":"2024-12-18T06:39:27.773545Z","iopub.status.idle":"2024-12-18T06:39:35.259361Z","shell.execute_reply.started":"2024-12-18T06:39:27.773500Z","shell.execute_reply":"2024-12-18T06:39:35.257994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = np.zeros(len(XX))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:35.260820Z","iopub.execute_input":"2024-12-18T06:39:35.261161Z","iopub.status.idle":"2024-12-18T06:39:35.266946Z","shell.execute_reply.started":"2024-12-18T06:39:35.261129Z","shell.execute_reply":"2024-12-18T06:39:35.265801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred += pipeline1.predict(XX)\npred += pipeline2.predict(XX)\npred += pipeline3.predict(XX)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:39:35.281474Z","iopub.execute_input":"2024-12-18T06:39:35.281930Z","iopub.status.idle":"2024-12-18T06:39:35.597510Z","shell.execute_reply.started":"2024-12-18T06:39:35.281886Z","shell.execute_reply":"2024-12-18T06:39:35.595061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.DataFrame({'id': XX['id'], 'sii': np.round(pred/3)})\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T06:41:41.353624Z","iopub.execute_input":"2024-12-18T06:41:41.354088Z","iopub.status.idle":"2024-12-18T06:41:41.365644Z","shell.execute_reply.started":"2024-12-18T06:41:41.354051Z","shell.execute_reply":"2024-12-18T06:41:41.364431Z"}},"outputs":[],"execution_count":null}]}