{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nimport numpy as np \nimport pandas as pd \nimport os\nfrom sklearn.impute import KNNImputer \nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV\nfrom sklearn.metrics import mean_squared_error\nimport lightgbm as lgb\n\n\ncount = 0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        count +=1\n        if count >= 15:\n            break\n    if count >= 15:\n        break\n        \ndata_train = pd.read_csv(r'/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndata_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_train.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:47.112727Z","iopub.execute_input":"2024-12-20T06:37:47.113711Z","iopub.status.idle":"2024-12-20T06:37:47.630419Z","shell.execute_reply.started":"2024-12-20T06:37:47.113672Z","shell.execute_reply":"2024-12-20T06:37:47.629411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# убираем нулевые значения искомой метрики, строим график распределения по значениям\n\n\nfiltered_data = data_train[data_train['sii'].notna()].reset_index()\n\nsns.countplot(data = filtered_data, x = 'sii')\n\nplt.title('Распределение Sii')\nplt.xlabel('Значение Sii')\nplt.ylabel('Количество')\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:47.632396Z","iopub.execute_input":"2024-12-20T06:37:47.633322Z","iopub.status.idle":"2024-12-20T06:37:47.873161Z","shell.execute_reply.started":"2024-12-20T06:37:47.633275Z","shell.execute_reply":"2024-12-20T06:37:47.872055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# распределение по полу и возрасту\nsns.countplot(data = filtered_data, x = 'Basic_Demos-Age', hue = 'Basic_Demos-Sex' )\n\nplt.title('Age/Sex распределение')\nplt.xlabel('количетсво')\nplt.legend(['мужской','женский'], title = 'Пол')\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:47.874611Z","iopub.execute_input":"2024-12-20T06:37:47.875229Z","iopub.status.idle":"2024-12-20T06:37:48.274796Z","shell.execute_reply.started":"2024-12-20T06:37:47.875182Z","shell.execute_reply":"2024-12-20T06:37:48.273966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# еще распределие, со стакнутыми колонками\npivot_table = filtered_data.pivot_table(index = 'Basic_Demos-Age', columns = 'Basic_Demos-Sex', aggfunc = 'size',fill_value = 0)\npivot_table.head()\n\npivot_table.plot(kind = 'bar', stacked = True)\nplt.title('Еще распредение для наглядности')\nplt.legend(['мужской', 'женский'], title ='пол')\nplt.tight_layout()\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:48.277874Z","iopub.execute_input":"2024-12-20T06:37:48.278614Z","iopub.status.idle":"2024-12-20T06:37:48.752722Z","shell.execute_reply.started":"2024-12-20T06:37:48.278563Z","shell.execute_reply":"2024-12-20T06:37:48.751916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Убираем колонки, которых нет в тестовом дата сете, кроме целевой\ncolumns_to_drop = [col for col in data_train.columns if col not in data_test.columns]\nprint(columns_to_drop)\ncolumns_to_drop.remove('PCIAT-PCIAT_Total')\nfiltered_data.drop(columns = columns_to_drop, inplace = True)\n\n# строим график, где смотрим распределение по нулевым значениям оставшихся колонок. здесь в итоге выбираем целевые 35%\ngraph = filtered_data.isnull().sum().apply(lambda x: x/len(filtered_data['PCIAT-PCIAT_Total'])*100).sort_values(ascending = False)\ngraph.plot(kind = 'barh', figsize = (20,30))\nplt.gca().yaxis.set_tick_params(labelsize=18, pad=5)\nplt.tight_layout()\nplt.gca().grid()\nticks = [x*5 for x in range(1,21)]\nplt.gca().xaxis.set_ticks(ticks)\n\n# удаляем столбцы с более 35% NaN\nindexes_to_delete = graph.index[graph > 35].tolist()\nfiltered_data.drop(columns = indexes_to_delete, inplace = True)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:48.753828Z","iopub.execute_input":"2024-12-20T06:37:48.754138Z","iopub.status.idle":"2024-12-20T06:37:50.144751Z","shell.execute_reply.started":"2024-12-20T06:37:48.754107Z","shell.execute_reply":"2024-12-20T06:37:50.143667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# так же убираем ID т к решили не использовать данные акселерометра\ncolumns_to_drop_2 = ['index','id']\nfiltered_data.drop(columns = columns_to_drop_2, inplace = True)\n\n# преобразуем категорийные столбцы в числовые. В основном это времена года когда проводились тесты\ncolumn_to_label = filtered_data.select_dtypes(include = 'object').columns\n\nlabeled_data = filtered_data.copy()\nlabel_coder = LabelEncoder()\nfor col in column_to_label:\n    labeled_data[f'{col}'] = label_coder.fit_transform(labeled_data[f'{col}'])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:50.146182Z","iopub.execute_input":"2024-12-20T06:37:50.146505Z","iopub.status.idle":"2024-12-20T06:37:50.161525Z","shell.execute_reply.started":"2024-12-20T06:37:50.146474Z","shell.execute_reply":"2024-12-20T06:37:50.160335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# строим матрицу в sns, анализируем на глаз, выбираем число 70%, убираем колонки с большей кореляцией(в основном это различные показатели массы и состава тела)\nmatrix = labeled_data.corr()\nplt.figure(figsize=(40, 40))\nsns.heatmap(matrix, annot=True, cmap='coolwarm', linewidths=0.5, fmt='.2f', vmin=-1, vmax=1)\n\nlimit = 0.7\n\ndrop_columns_3 = set()\nfor i in range(len(matrix.columns)):\n    for j in range(i):\n        if abs(matrix.iloc[i, j]) > limit:\n            col = matrix.columns[i]\n            drop_columns_3.add(col)\n\ndrop_columns_3.discard('PCIAT-PCIAT_Total')\n\ncleaned_data = labeled_data.drop(columns=drop_columns_3)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:50.162853Z","iopub.execute_input":"2024-12-20T06:37:50.163322Z","iopub.status.idle":"2024-12-20T06:37:56.830053Z","shell.execute_reply.started":"2024-12-20T06:37:50.163282Z","shell.execute_reply":"2024-12-20T06:37:56.829008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#заполняем пустые значения\nimputers = {}\nimputed_dataset = {}\n\nimputers = KNNImputer(n_neighbors=17)\n    \nimputed_dataset = pd.DataFrame(\n    imputers.fit_transform(cleaned_data), \n    columns=cleaned_data.columns)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:56.831484Z","iopub.execute_input":"2024-12-20T06:37:56.831859Z","iopub.status.idle":"2024-12-20T06:37:57.795837Z","shell.execute_reply.started":"2024-12-20T06:37:56.831810Z","shell.execute_reply":"2024-12-20T06:37:57.795019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Создаем модель, подбираем значения. Некоторые значения для начала подсматриваем у других\n    \nSEED = 42\n\n\nX = imputed_dataset.drop(columns=['PCIAT-PCIAT_Total'])  # your target column name\ny = imputed_dataset['PCIAT-PCIAT_Total']\n\n# Split data\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=SEED)\n'''\n# Parameter grid (potentially smaller)\nparam_grid = {\n    'learning_rate': [0.01, 0.03, 0.05, 0.1],\n    'max_depth': [6, 8, 10, 12, 14],\n    'num_leaves': [500],\n    'min_data_in_leaf': [5, 10, 20, 50],\n    'feature_fraction': [0.6, 0.8, 1.0],\n    'bagging_fraction': [0.6, 0.8, 0.9],\n    'bagging_freq': [1, 3, 5],\n    'lambda_l1': [4.735462555910575],\n    'lambda_l2': [4.735028557007343e-06],\n}\n\nmodel = lgb.LGBMRegressor(random_state=42, verbose=-1)\n\nsearch = RandomizedSearchCV(\n    estimator=model,\n    param_distributions=param_grid,\n    n_iter=3000,\n    scoring='neg_mean_squared_error',\n    cv=5,\n    random_state=42,\n    verbose=1\n)\n\nsearch.fit(X_train, y_train)\nprint(\"Best Parameters:\", search.best_params_)\nprint(f\"Best RMSE: {(-search.best_score_)**0.5:.4f}\")\n'''","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:57.799525Z","iopub.execute_input":"2024-12-20T06:37:57.799827Z","iopub.status.idle":"2024-12-20T06:37:57.812353Z","shell.execute_reply.started":"2024-12-20T06:37:57.799798Z","shell.execute_reply":"2024-12-20T06:37:57.811324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Еще один уточненный прогон\nparam_grid = {\n    'learning_rate': [0.05],  \n    'max_depth': [4, 6],\n    'num_leaves': [200, 300, 400, 500],\n    'min_data_in_leaf': [10],  \n    'feature_fraction': [0.4, 0.5, 0.6],\n    'bagging_fraction': [0.9],\n    'bagging_freq': [3],  \n    'lambda_l1': [1, 2, 3, 4, 4.735462555910575],\n    'lambda_l2': [3e-05, 3e-04, 4.735028557007343e-06],\n}\n\n\nmodel = lgb.LGBMRegressor(random_state=42, verbose=-1)\n\nsearch = RandomizedSearchCV(\n    estimator=model,\n    param_distributions=param_grid,\n    n_iter=10,\n    scoring='neg_mean_squared_error',\n    cv=5,\n    random_state=42,\n    verbose=1\n)\n\nsearch.fit(X_train, y_train)\nprint(\"Best Parameters:\", search.best_params_)\nprint(f\"Best RMSE: {(-search.best_score_)**0.5:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:37:57.814845Z","iopub.execute_input":"2024-12-20T06:37:57.815723Z","iopub.status.idle":"2024-12-20T06:38:02.656838Z","shell.execute_reply.started":"2024-12-20T06:37:57.815691Z","shell.execute_reply":"2024-12-20T06:38:02.655733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Формируем прогноз\nbest_model = search.best_estimator_\ncolumns_final = list(imputed_dataset.columns)\ncolumns_final.remove('PCIAT-PCIAT_Total')\nX_test = data_test[columns_final]\n\n# Select columns with object (categorical) data type\ncolumn_to_label_2 = X_test.select_dtypes(include='object').columns\n\n# Transform categorical columns using label_coder\nfor col in column_to_label_2:\n    X_test.loc[:, col] = label_coder.fit_transform(X_test[col])\n\nX_final_test = pd.DataFrame(\n    imputers.fit_transform(X_test), \n    columns=X_test.columns)\n\ny_pred = best_model.predict(X_final_test)\nX_final_test['PCIAT_TOTAL'] = y_pred\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:38:02.658449Z","iopub.execute_input":"2024-12-20T06:38:02.658860Z","iopub.status.idle":"2024-12-20T06:38:02.686747Z","shell.execute_reply.started":"2024-12-20T06:38:02.658814Z","shell.execute_reply":"2024-12-20T06:38:02.685700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Сохраняем прогноз\nX_final_test['sii'] = np.where(X_final_test['PCIAT_TOTAL']<21,0,\n                              np.where(X_final_test['PCIAT_TOTAL']<31,1,2))\nX_final_test['id'] = data_test['id']\nX_final_test[['id','sii']].to_csv('submission.csv', index=False,na_rep='null')\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T06:38:02.688121Z","iopub.execute_input":"2024-12-20T06:38:02.688557Z","iopub.status.idle":"2024-12-20T06:38:02.697448Z","shell.execute_reply.started":"2024-12-20T06:38:02.688511Z","shell.execute_reply":"2024-12-20T06:38:02.696605Z"}},"outputs":[],"execution_count":null}]}