{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9570255,"sourceType":"datasetVersion","datasetId":5833349}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns \nimport matplotlib.pyplot as plt\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import MinMaxScaler , StandardScaler , RobustScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import make_scorer , cohen_kappa_score , mean_squared_error as mse , f1_score as f1\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport optuna\nfrom sklearn.ensemble import BaggingClassifier , RandomForestClassifier\nfrom sklearn.base import clone\nfrom xgboost import XGBClassifier\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\npd.set_option(\"display.max_columns\" , 200)\npd.set_option(\"display.max_columns\" , 100)\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\ntrain_csv_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\"\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:25.732303Z","iopub.execute_input":"2024-11-02T13:26:25.732656Z","iopub.status.idle":"2024-11-02T13:26:28.378679Z","shell.execute_reply.started":"2024-11-02T13:26:25.732607Z","shell.execute_reply":"2024-11-02T13:26:28.377634Z"},"trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv\")\ndata_dict","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.380136Z","iopub.execute_input":"2024-11-02T13:26:28.380852Z","iopub.status.idle":"2024-11-02T13:26:28.407824Z","shell.execute_reply.started":"2024-11-02T13:26:28.380804Z","shell.execute_reply":"2024-11-02T13:26:28.406309Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dict[\"Field\"].unique()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.413178Z","iopub.execute_input":"2024-11-02T13:26:28.413744Z","iopub.status.idle":"2024-11-02T13:26:28.426179Z","shell.execute_reply.started":"2024-11-02T13:26:28.413691Z","shell.execute_reply":"2024-11-02T13:26:28.425133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_stats( cols = None , dtype = None , data = None ):\n    if isinstance(cols , str):\n        cols = [cols]\n    stats = []\n    for col in cols:\n        if data[col].dtype in [\"object\" , \"category\"] or (dtype is not None and dtype in [\"object\" , \"category\"]):\n            counts = data[col].value_counts(sort=False ,dropna = False ).astype(str)\n            percentage = (data[col].value_counts(sort = False , dropna=False , normalize = True) *100).round(2).astype(str)\n            ready_stats = (counts + \" (\" + percentage + \"%\" + \")\").to_frame(name=f\"{col} counts(%)\")\n            stats.append(ready_stats)\n        else:\n            num_stats = data[col].describe().to_frame().transpose()\n            num_stats[\"missing\"] = data[col].isnull().sum()\n            num_stats.index.name = col\n            stats.append(num_stats)\n            \n    return pd.concat(stats)\n\n\ndef groups_to(groupby=None , to = \"sii\" , data = None):\n    stats = data.groupby([groupby])[to].value_counts().unstack()\n    percentage_of_age_group_by_sii = stats.div(stats.sum(axis=1) , axis = 0) *100\n\n    final_stats = stats.astype(str) + \" (\" + percentage_of_age_group_by_sii.round(2).astype(str) + \"%\" \") \"\n    return final_stats","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.427936Z","iopub.execute_input":"2024-11-02T13:26:28.428681Z","iopub.status.idle":"2024-11-02T13:26:28.443987Z","shell.execute_reply.started":"2024-11-02T13:26:28.428633Z","shell.execute_reply":"2024-11-02T13:26:28.442012Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings \nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.449262Z","iopub.execute_input":"2024-11-02T13:26:28.450232Z","iopub.status.idle":"2024-11-02T13:26:28.456448Z","shell.execute_reply.started":"2024-11-02T13:26:28.450183Z","shell.execute_reply":"2024-11-02T13:26:28.455158Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\").drop(\"id\" , axis = 1)\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\").drop(\"id\" , axis = 1)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.458942Z","iopub.execute_input":"2024-11-02T13:26:28.459901Z","iopub.status.idle":"2024-11-02T13:26:28.653933Z","shell.execute_reply.started":"2024-11-02T13:26:28.459845Z","shell.execute_reply":"2024-11-02T13:26:28.652762Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_with_sii = train[train[\"sii\"].notna()][list(set(train.columns) - set(test.columns))]\ntrain_with_sii[train_with_sii.isna().any(axis=1)].head()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.655828Z","iopub.execute_input":"2024-11-02T13:26:28.656259Z","iopub.status.idle":"2024-11-02T13:26:28.701417Z","shell.execute_reply.started":"2024-11-02T13:26:28.656212Z","shell.execute_reply":"2024-11-02T13:26:28.700054Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_cols = [f\"PCIAT-PCIAT_{i:02d}\" for i in range(1 , 21)]\npciat_stuff = train_with_sii[pciat_cols + [\"PCIAT-PCIAT_Total\"]]\npciat_stuff","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.702863Z","iopub.execute_input":"2024-11-02T13:26:28.703216Z","iopub.status.idle":"2024-11-02T13:26:28.768235Z","shell.execute_reply.started":"2024-11-02T13:26:28.703181Z","shell.execute_reply":"2024-11-02T13:26:28.767004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12 , 5))\nsns.histplot(data=pciat_stuff , x=\"PCIAT-PCIAT_Total\" , bins=100)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:28.769422Z","iopub.execute_input":"2024-11-02T13:26:28.769737Z","iopub.status.idle":"2024-11-02T13:26:29.259272Z","shell.execute_reply.started":"2024-11-02T13:26:28.769704Z","shell.execute_reply":"2024-11-02T13:26:29.258057Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_stats(\"PCIAT-PCIAT_Total\" , data = pciat_stuff)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.260953Z","iopub.execute_input":"2024-11-02T13:26:29.261431Z","iopub.status.idle":"2024-11-02T13:26:29.282908Z","shell.execute_reply.started":"2024-11-02T13:26:29.261382Z","shell.execute_reply":"2024-11-02T13:26:29.281687Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_stuff[pciat_stuff[\"PCIAT-PCIAT_Total\"]==0].shape","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.284716Z","iopub.execute_input":"2024-11-02T13:26:29.285515Z","iopub.status.idle":"2024-11-02T13:26:29.295062Z","shell.execute_reply.started":"2024-11-02T13:26:29.285465Z","shell.execute_reply":"2024-11-02T13:26:29.294039Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"any_nun = pciat_stuff[pciat_stuff.isna().any(axis = 1)]\n(any_nun[pciat_cols].fillna(0).sum(axis = 1) != any_nun[\"PCIAT-PCIAT_Total\"]).sum()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.296386Z","iopub.execute_input":"2024-11-02T13:26:29.296833Z","iopub.status.idle":"2024-11-02T13:26:29.312691Z","shell.execute_reply.started":"2024-11-02T13:26:29.296781Z","shell.execute_reply":"2024-11-02T13:26:29.311402Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(pciat_stuff.fillna(0).sum(axis = 1) !=train_with_sii[\"PCIAT-PCIAT_Total\"])","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.314324Z","iopub.execute_input":"2024-11-02T13:26:29.314765Z","iopub.status.idle":"2024-11-02T13:26:29.331393Z","shell.execute_reply.started":"2024-11-02T13:26:29.314718Z","shell.execute_reply":"2024-11-02T13:26:29.330187Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_stuff.isnull().all(axis = 1).sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.332989Z","iopub.execute_input":"2024-11-02T13:26:29.334166Z","iopub.status.idle":"2024-11-02T13:26:29.343426Z","shell.execute_reply.started":"2024-11-02T13:26:29.334112Z","shell.execute_reply":"2024-11-02T13:26:29.342424Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an = train[ train[pciat_cols].notna().all(axis = 1)]\ntrain_an","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.344951Z","iopub.execute_input":"2024-11-02T13:26:29.345329Z","iopub.status.idle":"2024-11-02T13:26:29.472588Z","shell.execute_reply.started":"2024-11-02T13:26:29.345296Z","shell.execute_reply":"2024-11-02T13:26:29.471534Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_stats( cols = None , dtype = None , data = train_an ):\n    if isinstance(cols , str):\n        cols = [cols]\n    stats = []\n    for col in cols:\n        if data[col].dtype in [\"object\" , \"category\"] or (dtype is not None and dtype in [\"object\" , \"category\"]):\n            counts = data[col].value_counts(sort=False ,dropna = False ).astype(str)\n            percentage = (data[col].value_counts(sort = False , dropna=False , normalize = True) *100).round(2).astype(str)\n            ready_stats = (counts + \" (\" + percentage + \"%\" + \")\").to_frame(name=f\"{col} counts(%)\")\n            stats.append(ready_stats)\n        else:\n            num_stats = data[col].describe().to_frame().transpose()\n            num_stats[\"missing\"] = data[col].isnull().sum()\n            num_stats.index.name = col\n            stats.append(num_stats)\n            \n    return pd.concat(stats)\n\ndef groups_to(groupby=None , to = \"sii\" , data = train_an):\n    stats = data.groupby([groupby])[to].value_counts().unstack()\n    percentage_of_age_group_by_sii = stats.div(stats.sum(axis=1) , axis = 0) *100\n\n    final_stats = stats.astype(str) + \" (\" + percentage_of_age_group_by_sii.round(2).astype(str) + \"%\" \") \"\n    return final_stats","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.473910Z","iopub.execute_input":"2024-11-02T13:26:29.474286Z","iopub.status.idle":"2024-11-02T13:26:29.485702Z","shell.execute_reply.started":"2024-11-02T13:26:29.474250Z","shell.execute_reply":"2024-11-02T13:26:29.484408Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an[\"age_cat\"] = pd.cut(train_an[\"Basic_Demos-Age\"],\n                            bins = [4 , 12 , 18 , 22],\n                            labels=['Children (5-12)', 'Adolescents (13-18)', 'Adults (19-22)']\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.487079Z","iopub.execute_input":"2024-11-02T13:26:29.487421Z","iopub.status.idle":"2024-11-02T13:26:29.497722Z","shell.execute_reply.started":"2024-11-02T13:26:29.487381Z","shell.execute_reply":"2024-11-02T13:26:29.496632Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"groups_to(groupby = \"age_cat\" , to = \"sii\" , data = train_an)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.498955Z","iopub.execute_input":"2024-11-02T13:26:29.499330Z","iopub.status.idle":"2024-11-02T13:26:29.528835Z","shell.execute_reply.started":"2024-11-02T13:26:29.499289Z","shell.execute_reply":"2024-11-02T13:26:29.527902Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an[\"Sex\"] = train_an[\"Basic_Demos-Sex\"].map({0:\"Male\" , 1:\"Female\"})","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.530244Z","iopub.execute_input":"2024-11-02T13:26:29.530658Z","iopub.status.idle":"2024-11-02T13:26:29.538141Z","shell.execute_reply.started":"2024-11-02T13:26:29.530603Z","shell.execute_reply":"2024-11-02T13:26:29.536888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_map = {0: '< 1h/day', 1: '~ 1h/day', 2: '~ 2hs/day', 3: '> 3hs/day'}\ntrain_an[\"Internet_usage\"] = train_an[\"PreInt_EduHx-computerinternet_hoursday\"]\\\n.map(param_map).fillna(\"missing\")\n\nstats = get_stats(data = train_an , cols = \"Internet_usage\")\nstats ","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.548568Z","iopub.execute_input":"2024-11-02T13:26:29.549380Z","iopub.status.idle":"2024-11-02T13:26:29.567283Z","shell.execute_reply.started":"2024-11-02T13:26:29.549336Z","shell.execute_reply":"2024-11-02T13:26:29.566135Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_autopct(values):\n    def my_autopct(pct):\n        total = sum(values)\n        val = int(round(pct*total/100.0))\n        return f\"{val} ({pct.round(1)}% )\"\n    return my_autopct","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.570590Z","iopub.execute_input":"2024-11-02T13:26:29.570944Z","iopub.status.idle":"2024-11-02T13:26:29.578278Z","shell.execute_reply.started":"2024-11-02T13:26:29.570910Z","shell.execute_reply":"2024-11-02T13:26:29.577224Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(18, 10))\n\nstats_counts = stats[\"Internet_usage counts(%)\"]\\\n.str.split(\" \").str[0].astype(int)\n\nax1 = fig.add_subplot(2 , 3 , 1)\nsns.barplot(x=stats_counts.index, y=stats_counts.values \n            , ax = ax1, palette=\"Set3\")\n\nfor p , annotation in zip(ax1.patches , stats.values):\n    ax1.annotate(annotation[0], (p.get_x() + p.get_width() / 2., p.get_height())\n                 , ha ='center' , va='baseline', fontsize=10, color='black', xytext=(0, 5), \n                 textcoords='offset points')\n    \n    \nax2 = fig.add_subplot(2 , 3 , 2)\n    \n    \nsns.boxplot(data = train_an , y = \"Basic_Demos-Age\" , x = \"Internet_usage\" , ax = ax2 , palette = \"Set3\")\n\n\n\nfor i ,age_cat in zip(range(4, 7) , train_an.age_cat.unique().to_list()):\n    \n    ax = fig.add_subplot(2 , 3 , i )\n    \n    group = train_an.groupby(\"age_cat\")[\"Internet_usage\"].value_counts().unstack().T[age_cat]\n    \n    group.name = None\n    \n    group.plot(kind = \"pie\" \n                , legend = False\n                 ,autopct = make_autopct(group)\n                 ,ax = ax\n                 ,title = age_cat)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:29.579477Z","iopub.execute_input":"2024-11-02T13:26:29.579901Z","iopub.status.idle":"2024-11-02T13:26:30.581724Z","shell.execute_reply.started":"2024-11-02T13:26:29.579859Z","shell.execute_reply":"2024-11-02T13:26:30.580744Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.groupby(\"Sex\")[\"Internet_usage\"].value_counts().unstack().T[\"Male\"]","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:30.582992Z","iopub.execute_input":"2024-11-02T13:26:30.583310Z","iopub.status.idle":"2024-11-02T13:26:30.598240Z","shell.execute_reply.started":"2024-11-02T13:26:30.583278Z","shell.execute_reply":"2024-11-02T13:26:30.597423Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize = (15 , 5))\nfor i ,Sex in zip([1 , 2] , train_an[\"Sex\"].unique()):\n    \n    ax = fig.add_subplot(1 , 3 , i )\n    \n    group = train_an.groupby(\"Sex\")[\"Internet_usage\"].value_counts().unstack().T[Sex]\n    \n    group.name = None\n    \n    group.plot(kind = \"pie\" \n                , legend = False\n                 ,autopct = make_autopct(group)\n                 ,ax = ax\n                 ,title = Sex)\n    \nax = fig.add_subplot(1 , 3 , 3 )\nsns.countplot(data = train_an , x = \"Sex\")\nfor p in ax.patches :\n    ax.annotate(p.get_height() , (p.get_x() +p.get_width()/2 , p.get_height()),\n               ha='center' , va = \"baseline\", fontsize=10, color='black', xytext=(0, 5), \n                 textcoords='offset points')","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:30.599360Z","iopub.execute_input":"2024-11-02T13:26:30.599676Z","iopub.status.idle":"2024-11-02T13:26:31.109139Z","shell.execute_reply.started":"2024-11-02T13:26:30.599644Z","shell.execute_reply":"2024-11-02T13:26:31.108043Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_stats([\"CGAS-Season\" , \"CGAS-CGAS_Score\"] ).fillna(\"Not Available\")","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:31.110620Z","iopub.execute_input":"2024-11-02T13:26:31.111398Z","iopub.status.idle":"2024-11-02T13:26:31.135735Z","shell.execute_reply.started":"2024-11-02T13:26:31.111342Z","shell.execute_reply":"2024-11-02T13:26:31.134801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an[\"CGAS-Season\"].value_counts(dropna=False , sort = False ,)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:31.137045Z","iopub.execute_input":"2024-11-02T13:26:31.137428Z","iopub.status.idle":"2024-11-02T13:26:31.145493Z","shell.execute_reply.started":"2024-11-02T13:26:31.137394Z","shell.execute_reply":"2024-11-02T13:26:31.144527Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , ax = plt.subplots(1 , 3 ,figsize = (18, 5))\ntrain_an[\"CGAS-Season\"] = train_an[\"CGAS-Season\"].fillna(\"not_attended\")\ntrain_an[\"CGAS-Season\"].value_counts( sort = False).plot(kind=\"pie\",\n                                           autopct=make_autopct(train_an[\"CGAS-Season\"].value_counts( sort = False))\n                                           ,ax = ax[0])\n\nax[0].set_title(\"Attendence Season Distribution\")\n\nsns.histplot(data = train_an.query(\"Sex=='Male'\") , x = \"CGAS-CGAS_Score\" \n             , kde =True \n             ,ax = ax[1]\n            )\n\nax[1].set_title(\"Score Distribution For Males\")\n\n\nsns.histplot(data = train_an.query(\"Sex=='Female'\") , x = \"CGAS-CGAS_Score\" \n             , kde =True \n             ,ax = ax[2]\n            )\n\nax[2].set_title(\"Score Distribution For Females\")","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:31.146602Z","iopub.execute_input":"2024-11-02T13:26:31.146902Z","iopub.status.idle":"2024-11-02T13:26:32.022578Z","shell.execute_reply.started":"2024-11-02T13:26:31.146872Z","shell.execute_reply":"2024-11-02T13:26:32.021550Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = [\n    \"1-10: Needs constant supervision (24 hour care)\",\n    \"11-20: Needs considerable supervision\",\n    \"21-30: Unable to function in almost all areas\",\n    \"31-40: Major impairment in functioning in several areas\",\n    \"41-50: Moderate degree of interference in functioning\",\n    \"51-60: Variable functioning with sporadic difficulties\",\n    \"61-70: Some difficulty in a single area\",\n    \"71-80: No more than slight impairment in functioning\",\n    \"81-90: Good functioning in all areas\",\n    \"91-100: Superior functioning\"\n]\nbins = np.arange(0 , 101 , 10)\n\ntrain_an[\"CGAS_SCORE_CAT\"] = pd.cut(train_an[\"CGAS-CGAS_Score\"] , bins = bins , labels = labels)\n\ntrain[\"CGAS_SCORE_CAT\"] = pd.cut(train[\"CGAS-CGAS_Score\"] , bins = bins , labels = labels)\n\n\ncounts = train_an[\"CGAS_SCORE_CAT\"].value_counts(sort = False)\npercent = counts.div(counts.sum())*100\n\nfor_= counts.astype(str) + \" (\" + percent.round(2).astype(str) + \"%\" + \") \"\n\nfig = plt.figure(figsize=(12 , 12))\nax1 = fig.add_subplot(3 , 1 , 1)\nbars = ax1.barh(labels , counts)\n\nfor bar , label in zip(bars , for_):\n    plt.text(\n            bar.get_width() , bar.get_y() + bar.get_height()/2 , label, va= \"center\" )\n    \ngroup = train_an.groupby(\"CGAS_SCORE_CAT\")[\"sii\"]\n\nfor (label , group) , i in zip(group , range(6 , 16)):\n    ax = fig.add_subplot(3 , 5 , i)\n    \n    group_value_counts = group.value_counts()\n    \n    if group_value_counts.sum() ==0:\n        ax.set_visible(False)\n    \n    else:\n        \n        group_value_counts.plot(kind = \"pie\"\n                                 ,autopct = make_autopct(group_value_counts) \n                               ,ax = ax)\n\n        ax.set_title(label.split(\":\")[0])\n# plt.gca().invert_yaxis()\n# plt.grid()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:32.023772Z","iopub.execute_input":"2024-11-02T13:26:32.024131Z","iopub.status.idle":"2024-11-02T13:26:33.167654Z","shell.execute_reply.started":"2024-11-02T13:26:32.024085Z","shell.execute_reply":"2024-11-02T13:26:33.166618Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:33.168998Z","iopub.execute_input":"2024-11-02T13:26:33.169370Z","iopub.status.idle":"2024-11-02T13:26:33.299865Z","shell.execute_reply.started":"2024-11-02T13:26:33.169327Z","shell.execute_reply":"2024-11-02T13:26:33.298832Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.rename(\n    columns = {\"Physical-Height\":\"Height\" \n               , \"Physical-Weight\":\"Weight\"\n               ,\"Physical-BMI\":\"BMI\"\n               ,\"Physical-Waist_Circumference\" :\"Waist\"},\n    inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:33.301365Z","iopub.execute_input":"2024-11-02T13:26:33.301796Z","iopub.status.idle":"2024-11-02T13:26:33.308170Z","shell.execute_reply.started":"2024-11-02T13:26:33.301750Z","shell.execute_reply":"2024-11-02T13:26:33.307001Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Pcols = [\"Height\"  \n         , \"Weight\" \n         , \"BMI\"\n         , \"Waist\"]","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:33.309686Z","iopub.execute_input":"2024-11-02T13:26:33.310110Z","iopub.status.idle":"2024-11-02T13:26:33.317668Z","shell.execute_reply.started":"2024-11-02T13:26:33.310066Z","shell.execute_reply":"2024-11-02T13:26:33.316726Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_stats(Pcols)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:33.318767Z","iopub.execute_input":"2024-11-02T13:26:33.319074Z","iopub.status.idle":"2024-11-02T13:26:33.353169Z","shell.execute_reply.started":"2024-11-02T13:26:33.319043Z","shell.execute_reply":"2024-11-02T13:26:33.352267Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.loc[train_an[\"Weight\"]==0 ,\"Weight\"] = np.nan\ntrain_an.loc[train_an[\"BMI\"]==0 ,\"BMI\"] = np.nan","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:33.354453Z","iopub.execute_input":"2024-11-02T13:26:33.354800Z","iopub.status.idle":"2024-11-02T13:26:33.362070Z","shell.execute_reply.started":"2024-11-02T13:26:33.354764Z","shell.execute_reply":"2024-11-02T13:26:33.360938Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , ax = plt.subplots(2 , 3 , figsize=(16 , 10))\n\nhas_sii_higher_than_0 = [\"SII>0\" if x>0 else \"SII=0\" for x in train_an[\"sii\"]]\nhas_high_sii = [\"SII==3\" if x==3 else \"SII<3\" for x in train_an[\"sii\"]]\n\ncustom_palette = sns.color_palette([\"#B0C4DE\", \"#FF4500\"])\n\nsns.scatterplot(data=train_an \n                , x=\"Height\" \n                , y = \"Weight\" \n                , ax = ax[0,0] , hue = has_sii_higher_than_0 \n                , palette = \"Set1\")\n\nsns.scatterplot(data=train_an \n                , x=\"Weight\" \n                , y=\"BMI\" \n                , ax = ax[0,1] , hue=has_sii_higher_than_0\n                , palette = \"Set1\")\n\n\nsns.scatterplot(data=train_an \n                , x=\"Height\" \n                , y=\"BMI\" \n                , ax = ax[0,2] , hue=has_sii_higher_than_0\n                , palette = \"Set1\")\n\n\n\n\nsns.scatterplot(data=train_an \n                , x=\"Height\" \n                , y = \"Weight\" \n                , ax = ax[1,0] , hue = has_high_sii \n                , palette = custom_palette)\n\nsns.scatterplot(data=train_an \n                , x=\"Weight\" \n                , y=\"BMI\" \n                , ax = ax[1,1] , hue=has_high_sii\n                , palette = custom_palette)\n\n\nsns.scatterplot(data=train_an \n                , x=\"Height\" \n                , y=\"BMI\" \n                , ax = ax[1,2] , hue=has_high_sii\n                , palette = custom_palette)\n\nfor ax_ in ax.flatten():\n    ax_.grid(True)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:33.363520Z","iopub.execute_input":"2024-11-02T13:26:33.363903Z","iopub.status.idle":"2024-11-02T13:26:36.272181Z","shell.execute_reply.started":"2024-11-02T13:26:33.363860Z","shell.execute_reply":"2024-11-02T13:26:36.271165Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , axs = plt.subplots(1 , 3 , figsize=(16 , 5))\n\nfor col , ax in zip([\"Weight\" , \"Height\" , \"BMI\"] , axs) :\n    sns.histplot(data=train_an \n                 , x=col \n                 , ax = ax \n                 , hue = \"age_cat\" \n                 , palette=\"Set1\" , multiple=\"stack\")\n    \n    ax.grid(True)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:36.273430Z","iopub.execute_input":"2024-11-02T13:26:36.273737Z","iopub.status.idle":"2024-11-02T13:26:37.888501Z","shell.execute_reply.started":"2024-11-02T13:26:36.273705Z","shell.execute_reply":"2024-11-02T13:26:37.887457Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.rename(\n                columns = {\"Fitness_Endurance-Season\":\"Fittness_Season\"\n                          ,\"Fitness_Endurance-Max_Stage\" :\"Max_Stage\"\n                          ,\"Fitness_Endurance-Time_Mins\":\"Time(Mins)\"\n                          ,\"Fitness_Endurance-Time_Sec\":\"Time(Secs)\"},\n                inplace=True\n                )\nfittness_cols = [\"Max_Stage\" , \"Time(Mins)\" , \"Time(Secs)\"]\ntrain_an","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:37.889925Z","iopub.execute_input":"2024-11-02T13:26:37.890370Z","iopub.status.idle":"2024-11-02T13:26:38.026650Z","shell.execute_reply.started":"2024-11-02T13:26:37.890321Z","shell.execute_reply":"2024-11-02T13:26:38.025619Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an[fittness_cols + [\"Fittness_Season\"]].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:38.027922Z","iopub.execute_input":"2024-11-02T13:26:38.028280Z","iopub.status.idle":"2024-11-02T13:26:38.038435Z","shell.execute_reply.started":"2024-11-02T13:26:38.028245Z","shell.execute_reply":"2024-11-02T13:26:38.037438Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.loc[\n    train_an[[\"Time(Secs)\", \"Time(Mins)\"]].isnull().any(axis=1) &\n    ~train_an[[\"Time(Secs)\", \"Time(Mins)\"]].isnull().all(axis=1), \n    fittness_cols\n] #index in [420 , 2907]\n","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:38.039680Z","iopub.execute_input":"2024-11-02T13:26:38.040046Z","iopub.status.idle":"2024-11-02T13:26:38.057946Z","shell.execute_reply.started":"2024-11-02T13:26:38.040006Z","shell.execute_reply":"2024-11-02T13:26:38.056903Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.loc[train_an.index==420 , \"Time(Secs)\"] = 0\ntrain_an.loc[train_an.index==2907 , \"Time(Secs)\"] = np.nan","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:38.059098Z","iopub.execute_input":"2024-11-02T13:26:38.059421Z","iopub.status.idle":"2024-11-02T13:26:38.068959Z","shell.execute_reply.started":"2024-11-02T13:26:38.059380Z","shell.execute_reply":"2024-11-02T13:26:38.068037Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_stats(fittness_cols)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:38.070217Z","iopub.execute_input":"2024-11-02T13:26:38.070717Z","iopub.status.idle":"2024-11-02T13:26:38.102136Z","shell.execute_reply.started":"2024-11-02T13:26:38.070677Z","shell.execute_reply":"2024-11-02T13:26:38.101105Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.Max_Stage.unique()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:38.103510Z","iopub.execute_input":"2024-11-02T13:26:38.103832Z","iopub.status.idle":"2024-11-02T13:26:38.111489Z","shell.execute_reply.started":"2024-11-02T13:26:38.103798Z","shell.execute_reply":"2024-11-02T13:26:38.110255Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(16 , 15))\naxes = []\nexplode = [0.1 , 0 , 0, 0]\nfor x in range(1 , 4):\n    ax = fig.add_subplot(3 , 3 ,x)\n    axes.append(ax)\n\nfor ax , col in zip(axes , fittness_cols):\n    sns.histplot(data = train_an \n                 , x=col , ax = ax\n                 , hue = has_sii_higher_than_0\n                 , multiple=\"stack\"\n                 , stat=\"count\" )\n    ax.grid(True)\n    \nax3 , ax4 , ax5 = [fig.add_subplot(3 , 3 , x) for x in range(4 , 7)]\n\nsii_counts_for_max_stage = train_an.loc[train_an[\"Max_Stage\"]>7 , \"sii\"].value_counts()\n\nsii_counts_for_max_stage.plot(\n                               kind=\"pie\",\n                               autopct=make_autopct(sii_counts_for_max_stage),\n                               ax = ax3,\n                               title=\"SII For Kids with Max Stage > 7\",\n                               explode=[0.1 , 0 , 0],\n                               shadow=True)\n\n\n\n\nsii_counts_for_max_stage = train_an.loc[train_an[\"Max_Stage\"]<=7 , \"sii\"].value_counts()\n\nsii_counts_for_max_stage.plot(\n                               kind=\"pie\",\n                               autopct=make_autopct(sii_counts_for_max_stage),\n                               ax = ax4,\n                               title=\"SII For Kids with Max Stage <= 7\",\n                               explode=explode,\n                               shadow=True)\n\ntrain_an[\"Total_Time\"] = train_an[\"Time(Mins)\"] + train_an[\"Time(Secs)\"]/60\n\nsns.boxplot(data = train_an \n            , x=\"sii\" \n            , y=\"Total_Time\" \n            , ax=ax5\n            , palette=\"Set3\")\nax5.grid(True)\n\nax6 = fig.add_subplot(3 , 1 , 3)\n\nseason_value_counts = train_an[\"Fittness_Season\"].value_counts()\nseason_value_counts.plot(kind=\"pie\",\n                       autopct=make_autopct(season_value_counts)\n                       ,ax=ax6\n                       ,title=\"Enrollment Season\"\n                        ,shadow=True , explode=[0.1,0.1,0.1,0.1])","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:38.113226Z","iopub.execute_input":"2024-11-02T13:26:38.113690Z","iopub.status.idle":"2024-11-02T13:26:40.340500Z","shell.execute_reply.started":"2024-11-02T13:26:38.113636Z","shell.execute_reply":"2024-11-02T13:26:40.339424Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.rename(\n                columns={\"BIA-BIA_Activity_Level_num\" : \"Activity_lvl\"},\n                inplace=True\n                )\n\ntrain_an","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:40.341802Z","iopub.execute_input":"2024-11-02T13:26:40.342161Z","iopub.status.idle":"2024-11-02T13:26:40.476809Z","shell.execute_reply.started":"2024-11-02T13:26:40.342124Z","shell.execute_reply":"2024-11-02T13:26:40.475847Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"groups = train_an.groupby(\"Activity_lvl\")[\"sii\"]\nfig , axes = plt.subplots(1 ,5 , figsize=(18 , 9))\nactivity_lvl_map = {1:\"Very Light\", 2:\"Light\", 3:\"Moderate\", 4:\"Heavy\", 5:\"Exceptional\"}\nfor ax , (activity_lvl , group) in zip(axes , groups):\n    value_counts = group.value_counts()\n    value_counts.name=None\n    value_counts.plot(kind=\"pie\" \n                      ,ax=ax\n                      ,autopct=make_autopct(value_counts)\n                      ,title=f\"SII for Activity Level : {activity_lvl_map[activity_lvl]}\")\n\n    ","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:40.478102Z","iopub.execute_input":"2024-11-02T13:26:40.478470Z","iopub.status.idle":"2024-11-02T13:26:41.115059Z","shell.execute_reply.started":"2024-11-02T13:26:40.478437Z","shell.execute_reply":"2024-11-02T13:26:41.113976Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12 , 5))\ntrain_an[\"Activity_lvl_mapped\"] = train_an[\"Activity_lvl\"].map(activity_lvl_map)\nsns.countplot(data = train_an , x=\"age_cat\" , hue = \"Activity_lvl_mapped\")","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:41.116457Z","iopub.execute_input":"2024-11-02T13:26:41.116789Z","iopub.status.idle":"2024-11-02T13:26:41.470002Z","shell.execute_reply.started":"2024-11-02T13:26:41.116755Z","shell.execute_reply":"2024-11-02T13:26:41.468944Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.groupby(\"Activity_lvl_mapped\")[\"BMI\"]","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:41.471535Z","iopub.execute_input":"2024-11-02T13:26:41.472307Z","iopub.status.idle":"2024-11-02T13:26:41.479166Z","shell.execute_reply.started":"2024-11-02T13:26:41.472257Z","shell.execute_reply":"2024-11-02T13:26:41.478243Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , axes = plt.subplots(1 , 5 , figsize = (25 , 8))\ngrouped = train_an.groupby(\"Activity_lvl_mapped\")[\"BMI\"]\n\nfor ax , (name , group) in zip(axes , grouped):\n    group.plot(kind=\"box\"\n               , ax = ax)\n    ax.set_title(f\"Activity Level : {name}\")\n    ","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:41.480611Z","iopub.execute_input":"2024-11-02T13:26:41.481082Z","iopub.status.idle":"2024-11-02T13:26:42.478304Z","shell.execute_reply.started":"2024-11-02T13:26:41.481037Z","shell.execute_reply":"2024-11-02T13:26:42.477340Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.groupby(\"Activity_lvl_mapped\" )\\\n[[\"BMI\" \n  ,\"Weight\" \n  ,\"Height\"]].agg([\"min\" , \"max\"])","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:42.479658Z","iopub.execute_input":"2024-11-02T13:26:42.479998Z","iopub.status.idle":"2024-11-02T13:26:42.503501Z","shell.execute_reply.started":"2024-11-02T13:26:42.479943Z","shell.execute_reply":"2024-11-02T13:26:42.502443Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:42.504777Z","iopub.execute_input":"2024-11-02T13:26:42.505130Z","iopub.status.idle":"2024-11-02T13:26:42.631793Z","shell.execute_reply.started":"2024-11-02T13:26:42.505096Z","shell.execute_reply":"2024-11-02T13:26:42.630669Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FGC = ['FGC-Season','FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone',\n       'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone',\n       'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n       'FGC-FGC_TL', 'FGC-FGC_TL_Zone']\n\nnum_cols = [col for col in FGC if (col.split(\"_\")[-1]!=\"Zone\") and (col != 'FGC-Season')]\nzone_cols = [col for col in FGC if col.split(\"_\")[-1]==\"Zone\"]\ndf = train_an[FGC]\ndf","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:42.633225Z","iopub.execute_input":"2024-11-02T13:26:42.633593Z","iopub.status.idle":"2024-11-02T13:26:42.672149Z","shell.execute_reply.started":"2024-11-02T13:26:42.633554Z","shell.execute_reply":"2024-11-02T13:26:42.670991Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[num_cols].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:42.673612Z","iopub.execute_input":"2024-11-02T13:26:42.674070Z","iopub.status.idle":"2024-11-02T13:26:42.685328Z","shell.execute_reply.started":"2024-11-02T13:26:42.674022Z","shell.execute_reply":"2024-11-02T13:26:42.684390Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[zone_cols].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:42.686927Z","iopub.execute_input":"2024-11-02T13:26:42.687942Z","iopub.status.idle":"2024-11-02T13:26:42.696226Z","shell.execute_reply.started":"2024-11-02T13:26:42.687906Z","shell.execute_reply":"2024-11-02T13:26:42.695219Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig , axes = plt.subplots(2 , 3 , figsize = (18 , 10))\ncurr_num = [\"FGC-FGC_PU\" , \"FGC-FGC_CU\" , \"FGC-FGC_TL\"]\ncurr_zone = [\"FGC-FGC_PU_Zone\" , \"FGC-FGC_CU_Zone\" , \"FGC-FGC_TL_Zone\"]\nflattened_axes_up = axes.flatten()[:3]\nflattened_axes_down = axes.flatten()[3:]\n\n\nfor ax , num , Zone in zip(flattened_axes_up , curr_num , curr_zone):\n    if num is None:\n        break\n    sns.boxplot(data = train_an \n                , x = Zone \n                , y = num \n                , ax = ax\n                ,palette=\"Set2\")\n    ax.grid(True)\n    \nfor ax , num in zip(flattened_axes_down , curr_num):\n    sns.boxplot(data = train_an \n                , x = \"age_cat\" \n                , y=num \n                , ax = ax \n                , palette=\"Set2\")\n    \n    ax.set_xticklabels( x.split(\" \")[-1] for x in train_an[\"age_cat\"].unique().to_list())\n    \n    \n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:42.697584Z","iopub.execute_input":"2024-11-02T13:26:42.697936Z","iopub.status.idle":"2024-11-02T13:26:43.792055Z","shell.execute_reply.started":"2024-11-02T13:26:42.697902Z","shell.execute_reply":"2024-11-02T13:26:43.791004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_an.groupby(\"FGC-FGC_PU_Zone\")[\"FGC-FGC_PU\"].agg([\"min\" , \"max\"])","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:43.793375Z","iopub.execute_input":"2024-11-02T13:26:43.793673Z","iopub.status.idle":"2024-11-02T13:26:43.806704Z","shell.execute_reply.started":"2024-11-02T13:26:43.793642Z","shell.execute_reply":"2024-11-02T13:26:43.805731Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_to_exclude = [\"Physical-Diastolic_BP\" , \"Physical-HeartRate\" , \"Physical-Systolic_BP\"\n                     ,\"Time(Secs)\" , \"Time(Mins)\" , \"Sex\" , \"Time(Mins)\" \n                      , \"Activity_lvl_mapped\" , \"Internet_usage\"] + pciat_cols + [\"sii\" , \"PCIAT-PCIAT_Total\"]\n\ncat_mapper ={\n    \"not_attended\":-1,\n    \"Spring\": 1,\n    \"Summer\": 2,\n    \"Fall\": 3,\n    \"Winter\": 4\n    ,np.nan:-1\n}\ncat_mapper","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:43.808149Z","iopub.execute_input":"2024-11-02T13:26:43.808563Z","iopub.status.idle":"2024-11-02T13:26:43.817511Z","shell.execute_reply.started":"2024-11-02T13:26:43.808518Z","shell.execute_reply":"2024-11-02T13:26:43.816600Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ny = train_an[\"PCIAT-PCIAT_Total\"]\n\ny_ = train_an[\"sii\"]\n\ndata = train_an.drop(columns_to_exclude , axis=1)\ndata\n\nfor col in data:\n    if \"Season\" in col:\n        data[col] = data[col].map(cat_mapper)\n    if data[col].dtype == \"category\":\n        data[col] = data[col].cat.codes\n        \n        \ndata","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:43.818802Z","iopub.execute_input":"2024-11-02T13:26:43.819234Z","iopub.status.idle":"2024-11-02T13:26:43.917207Z","shell.execute_reply.started":"2024-11-02T13:26:43.819186Z","shell.execute_reply":"2024-11-02T13:26:43.916211Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cols = [\"Basic_Demos-Sex\" , \"Basic_Demos-Sex\" , \"Weight\" \n#         , \"Height\" , \"FGC-FGC_CU\" , \"FGC-FGC_PU\" \n#         , \"Activity_lvl\" ,\"PAQ_A-PAQ_A_Total\" ,\"PAQ_C-PAQ_C_Total\"\n#        ,\"PreInt_EduHx-computerinternet_hoursday\" , \"SDS-SDS_Total_Raw\" , \"CGAS_SCORE_CAT\" ,\"Total_Time\"]\n\n# X_train , X_test , y_train , y_test = train_test_split(data , y_\n#                                                        , test_size = 0.2 \n#                                                        , random_state = 42 \n#                                                        , stratify = train_an[\"sii\"])","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:43.918803Z","iopub.execute_input":"2024-11-02T13:26:43.919130Z","iopub.status.idle":"2024-11-02T13:26:43.923612Z","shell.execute_reply.started":"2024-11-02T13:26:43.919098Z","shell.execute_reply":"2024-11-02T13:26:43.922557Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# baseline_pipeline = Pipeline(steps = [(\"impute\" , SimpleImputer(strategy=\"mean\" \n#                                                                 , add_indicator=True))\n#                                      ,(\"scale\" , MinMaxScaler())])\n\n# no = Pipeline(steps = [(\"impute\" , SimpleImputer(strategy=\"mean\" \n#                                                                 , add_indicator=True))])\n\n\n# X_train_pre = baseline_pipeline.fit_transform(X_train)\n# X_test_pre = baseline_pipeline.transform(X_test)\n# feat_dim = X_train_pre.shape[1]\n\n# X_train_pre_ = no.fit_transform(X_train)\n# X_test_pre_ = no.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:43.924891Z","iopub.execute_input":"2024-11-02T13:26:43.925263Z","iopub.status.idle":"2024-11-02T13:26:43.937488Z","shell.execute_reply.started":"2024-11-02T13:26:43.925227Z","shell.execute_reply":"2024-11-02T13:26:43.936590Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"l = []\nfor col in data:\n    non_num = [\"Season\" , \"Zone\" , \"Sex\" , \"BIA\"]\n    if all(x not in col for x in non_num):\n        l.append(col)\n\ndata[l].hist(bins = 20 , figsize = (20 , 20))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:43.938758Z","iopub.execute_input":"2024-11-02T13:26:43.939151Z","iopub.status.idle":"2024-11-02T13:26:48.787221Z","shell.execute_reply.started":"2024-11-02T13:26:43.939108Z","shell.execute_reply":"2024-11-02T13:26:48.786237Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data[l].plot(kind=\"box\" \n             , subplots = True\n             , layout = (5 , 5)\n             , figsize = (20 , 20))\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T13:26:48.788612Z","iopub.execute_input":"2024-11-02T13:26:48.788994Z","iopub.status.idle":"2024-11-02T13:26:51.845061Z","shell.execute_reply.started":"2024-11-02T13:26:48.788939Z","shell.execute_reply":"2024-11-02T13:26:51.844036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert(scores , scaling_fact = 1 , thresholds = [30 , 50 , 80]):\n    a , b ,c = thresholds\n    scores = np.array(scores)*scaling_fact\n    bins = np.zeros_like(scores)\n    bins[scores <= a] = 0\n    bins[(scores > a) & (scores < b)] = 1\n    bins[(scores >= b) & (scores < c)] = 2\n    bins[scores >= c] = 3\n    return bins\n\n\ndef quadratic_kappa(y_true, y_pred , scaling_fact = 1 , convert_=True , thresholds = [30 , 50 , 80]):\n    if convert_==True :\n        y_true = convert(y_true , scaling_fact = scaling_fact , thresholds = [30 , 50 , 80])\n        y_pred = convert(y_pred, scaling_fact = scaling_fact, thresholds = thresholds)\n        \n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\nkappa_scorer = make_scorer(quadratic_kappa , greater_is_better = True )\n\n\n\ndef cv(model , X , y , n_splits = 5 , random_state = 42 , scaling_fact=1 , oof=False , thresholds = [30 , 50 , 80]):\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n    \n    scores = []\n    oof_numeric = np.zeros(len(y))\n    oof_converted= np.zeros(len(y))\n    # Loop through each fold\n    for train_idx, val_idx in skf.split(X, y):\n        # Split the data into training and validation sets\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Clone the model to ensure each fold has an independent model\n        cloned_model = clone(model)\n        # Fit the model on the training set\n        cloned_model.fit(X_train, y_train)\n        \n        # Predict on the validation set\n        y_pred = cloned_model.predict(X_val)\n        score = quadratic_kappa(y_val , y_pred , scaling_fact=scaling_fact , thresholds = thresholds)\n        scores.append(score)\n        oof_numeric[val_idx] = y_pred\n        oof_converted[val_idx] = convert(y_pred , scaling_fact = scaling_fact, thresholds = thresholds)\n        \n    if not oof:\n        return np.array(scores)\n    else:\n        return np.array(scores), oof_numeric, oof_converted\n","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:51.846535Z","iopub.execute_input":"2024-11-02T13:26:51.846934Z","iopub.status.idle":"2024-11-02T13:26:51.863100Z","shell.execute_reply.started":"2024-11-02T13:26:51.846889Z","shell.execute_reply":"2024-11-02T13:26:51.862170Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# default_params = {\n#     'n_estimators': 284,\n#     'max_depth': 3,\n#     'learning_rate': 0.020744993080577038,\n#     'subsample': 0.5300922367461915,\n#     'colsample_bytree': 0.39933847867075173,\n#     'min_child_weight': 20,\n#     'gamma': 0.7647839523627029,\n#     'reg_alpha': 13.370914618159713,\n#     'reg_lambda': 7.437465259222301,\n#     'scale_pos_weight': 7.524346671537227,\n#     'max_bin': 419,\n#     'n_jobs': -1,\n#     'random_state': 25,\n#     'objective': 'reg:squarederror'\n# }\n\n# # Define a function to optimize using Optuna\n# def objective(trial):\n#     # Create a parameter dictionary based on default_params\n#     params = {\n#         **default_params,  # Start with default parameters\n#         'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n#         'max_depth': trial.suggest_int('max_depth', 1, 10),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-5, 0.1),\n#         'subsample': trial.suggest_uniform('subsample', 0.1, 1.0),\n#         'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.1, 1.0),\n#         'min_child_weight': trial.suggest_int('min_child_weight', 1, 50),\n#         'gamma': trial.suggest_loguniform('gamma', 1e-5, 10.0),\n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-5, 50.0),\n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-5, 50.0),\n#         'scale_pos_weight': trial.suggest_loguniform('scale_pos_weight', 1e-5, 50.0),\n#         'max_bin': trial.suggest_int('max_bin', 1, 1000)\n#     }\n\n#     scores , oof_num_xgb , oof_con = cv(xgb, new ,  y, n_splits = 10, scaling_fact=1.3 , oof = True)\n\n#     y_train_real = convert(y, scaling_fact=1)\n#     oof_real_xgb = convert(oof_num_xgb, scaling_fact=1.3)\n#     score = quadratic_kappa(y_train_real, oof_real_xgb, scaling_fact=1, convert_=False)\n    \n    \n#     return score  # Objective to minimize\n\n\n\n# study = optuna.create_study(direction='maximize')\n\n# # Optimize the objective function\n# study.optimize(objective, n_trials=400)  # Adjust the number of trials as needed","metadata":{"execution":{"iopub.status.busy":"2024-11-02T13:26:51.864570Z","iopub.execute_input":"2024-11-02T13:26:51.864936Z","iopub.status.idle":"2024-11-02T13:26:51.880810Z","shell.execute_reply.started":"2024-11-02T13:26:51.864901Z","shell.execute_reply":"2024-11-02T13:26:51.879947Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"whole_data =  pd.read_csv(train_csv_path)\n\npar_dir = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\"\n\nids = [id_.split(\"=\")[1] for id_ in os.listdir(par_dir)]\nlabel_for_ids = [whole_data.query(f\"id == '{id_}'\")[\"sii\"].values[0] for id_ in ids]\n\nlabel_for_ids_pciat = [whole_data.query(f\"id == '{id_}'\")[\"PCIAT-PCIAT_Total\"].values[0] for id_ in ids]\n\nid_files = [os.path.join(par_dir , id_ ,\"part-0.parquet\") for id_ in os.listdir(par_dir)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T13:26:51.881838Z","iopub.execute_input":"2024-11-02T13:26:51.882173Z","iopub.status.idle":"2024-11-02T13:27:27.258783Z","shell.execute_reply.started":"2024-11-02T13:26:51.882141Z","shell.execute_reply":"2024-11-02T13:27:27.257864Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\nimport random\ndef reduce_seq(seq_list : list[pd.DataFrame], samples : int = 30000 ) ->list[pd.DataFrame]:  #Randomly removing instances\n    return_list = []\n    for sequence in seq_list:\n        if sequence.shape[0]<samples:\n            padded_seq = pad_sequences([sequence] , maxlen=samples , padding=\"post\")\\\n            .reshape(samples , sequence.shape[1])\n            return_list.append(padded_seq)\n        else : \n            truncated_seq = sequence.sample(samples).sort_index().values\n            return_list.append(truncated_seq)\n\n    return np.array(return_list)\n            ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T13:27:27.259998Z","iopub.execute_input":"2024-11-02T13:27:27.260287Z","iopub.status.idle":"2024-11-02T13:27:30.467641Z","shell.execute_reply.started":"2024-11-02T13:27:27.260257Z","shell.execute_reply":"2024-11-02T13:27:30.466518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ndef get_timeseries_path(train = 5 , test = 1 , org_data = pd.read_csv(train_csv_path) , label = \"sii\"):\n    global label_for_ids , ids , id_files , label_for_ids_pciat\n\n    if label == \"pciat\":\n        label_for_ids = label_for_ids_pciat\n    \n    train_pathes , train_labels = id_files[0:train] , label_for_ids[0:train]\n    \n    test_pathes , test_labels = id_files[train : train+test] , label_for_ids[train : train+test] \n    \n    return (train_pathes , np.array(train_labels).reshape(-1 , 1) , test_pathes , np.array(test_labels ).reshape(-1 , 1)) #\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T13:27:30.476476Z","iopub.execute_input":"2024-11-02T13:27:30.477167Z","iopub.status.idle":"2024-11-02T13:27:30.531437Z","shell.execute_reply.started":"2024-11-02T13:27:30.477129Z","shell.execute_reply":"2024-11-02T13:27:30.530628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sampels , test_samples = 22 , 20\ntrain_seq , y_train_seq , test_seq , y_test_seq = get_timeseries_path(train=train_sampels , test = test_samples)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T14:59:37.837674Z","iopub.execute_input":"2024-11-02T14:59:37.838613Z","iopub.status.idle":"2024-11-02T14:59:37.843109Z","shell.execute_reply.started":"2024-11-02T14:59:37.838572Z","shell.execute_reply":"2024-11-02T14:59:37.842050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seq_list_train = [pd.read_parquet(x)for x in train_seq]\nseq_list_test = [pd.read_parquet(x)for x in test_seq]\n\nsamples_size = 20000\n\ntrain_seq_red = reduce_seq(seq_list_train , samples = samples_size)\ntest_seq_red = reduce_seq(seq_list_test , samples = samples_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T14:59:39.957138Z","iopub.execute_input":"2024-11-02T14:59:39.957909Z","iopub.status.idle":"2024-11-02T14:59:41.472115Z","shell.execute_reply.started":"2024-11-02T14:59:39.957869Z","shell.execute_reply":"2024-11-02T14:59:41.471251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#Get Data\ntrain_sampels , test_samples = 52 , 20\ntrain_seq , y_train_seq , test_seq , y_test_seq = get_timeseries_path(train=train_sampels , test = test_samples)\n\n\n#preprocess Data\nseq_list_train = [pd.read_parquet(x)for x in train_seq]\nseq_list_test = [pd.read_parquet(x)for x in test_seq]\n\nsamples_size = 20000\n\ntrain_seq_red = reduce_seq(seq_list_train , samples = samples_size)\ntest_seq_red = reduce_seq(seq_list_test , samples = samples_size)\n\n\n#Scale Data\nreshaped_data_train = train_seq_red.reshape(-1 , 13)\nreshaped_data_test= test_seq_red.reshape(-1 , 13)\n\nscaler = RobustScaler()\n\nscaled_data_train = scaler.fit_transform(reshaped_data_train)\nscaled_data_test = scaler.fit_transform(reshaped_data_test)\n\ntrain_seq_red = scaled_data_train.reshape(train_sampels , samples_size , 13)\ntest_seq_red = scaled_data_test.reshape(test_samples , samples_size , 13)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T15:05:44.623684Z","iopub.execute_input":"2024-11-02T15:05:44.624363Z","iopub.status.idle":"2024-11-02T15:05:49.570298Z","shell.execute_reply.started":"2024-11-02T15:05:44.624322Z","shell.execute_reply":"2024-11-02T15:05:49.569338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_tf(value , thresholds = (1 , 2 , 3)):\n    a, b, c = thresholds\n    return tf.where(value<a , 0 ,\n                tf.where((value>=a) & (value<b) ,1 ,\n                        tf.where((value>=b) & (value<c) , 2 ,3)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T13:58:11.123215Z","iopub.execute_input":"2024-11-02T13:58:11.124145Z","iopub.status.idle":"2024-11-02T13:58:11.131342Z","shell.execute_reply.started":"2024-11-02T13:58:11.124076Z","shell.execute_reply":"2024-11-02T13:58:11.130135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.losses import Loss\nfrom tensorflow.keras.metrics import MeanSquaredError as MSE, MeanAbsoluteError as MAE\n\nclass MohammedLoss(Loss):\n    def __init__(self, delta=1.0, **kwargs):\n        super().__init__(**kwargs)\n        self.delta = delta\n\n    def call(self, y_true, y_pred):\n        mae_value = tf.abs(y_true - y_pred)\n        mse_value = tf.square(y_true - y_pred)\n\n        # Use tf.where to conditionally return MSE or MAE\n        loss = tf.where(mae_value >= self.delta,  # Condition\n                         mse_value,  # True branch: return MSE\n                         mae_value)  # False branch: return MAE\n\n        return loss\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T14:01:05.037680Z","iopub.execute_input":"2024-11-02T14:01:05.038687Z","iopub.status.idle":"2024-11-02T14:01:05.047986Z","shell.execute_reply.started":"2024-11-02T14:01:05.038640Z","shell.execute_reply":"2024-11-02T14:01:05.047002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_tf(value , thresholds = (1 , 2 , 3)):\n    a, b, c = thresholds\n    return tf.where(value<a , 0 ,\n                tf.where((value>=a) & (value<b) ,1 ,\n                        tf.where((value>=b) & (value<c) , 2 ,3)))\n    \nclass MohammedLossQWK(Loss):\n    def __init__(self, thresholds=(1 , 2 , 3) , cons = 1., **kwargs):\n        super().__init__(**kwargs)\n        self.thresholds = thresholds\n        self.cons = cons\n\n    def call(self, y_true, y_pred):\n\n        y_true_rounded , y_pred_rounder = convert_tf(y_true , thresholds = self.thresholds) , convert_tf(y_pred , thresholds = self.thresholds)\n\n\n        loss_pen = tf.abs(y_true_rounded - y_pred_rounder) * self.cons\n\n\n\n        loss_mse = tf.square(y_true - y_pred)\n\n        return loss_mse + tf.cast(loss_pen , dtype = tf.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T14:25:16.560775Z","iopub.execute_input":"2024-11-02T14:25:16.561628Z","iopub.status.idle":"2024-11-02T14:25:16.570045Z","shell.execute_reply.started":"2024-11-02T14:25:16.561589Z","shell.execute_reply":"2024-11-02T14:25:16.569003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import Huber\n\n\nmodel = models.Sequential([\n    layers.LSTM(255, return_sequences=True , input_shape=(30000, 13)),\n    layers.LSTM(250, return_sequences=True),\n    layers.LSTM(50, return_sequences=False),\n    layers.Dense(1)                           \n])\n# Compile the model\nmodel.compile(optimizer=Adam(), loss=MohammedLoss(delta=0.1), metrics=['mae'])\n\n# Model summary\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-02T15:05:55.921198Z","iopub.execute_input":"2024-11-02T15:05:55.921621Z","iopub.status.idle":"2024-11-02T15:05:56.052312Z","shell.execute_reply.started":"2024-11-02T15:05:55.921581Z","shell.execute_reply":"2024-11-02T15:05:56.051251Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\nearly_stopping = EarlyStopping(monitor='val_loss'\n                               , patience=20 \n                               , restore_best_weights = True)\n\n\nmodel.fit(train_seq_red, y_train_seq\n          , epochs=20 \n          , batch_size = 52\n          , validation_data = (test_seq_red ,y_test_seq )\n          , callbacks = [early_stopping])  # Example labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T15:05:58.737067Z","iopub.execute_input":"2024-11-02T15:05:58.737472Z","iopub.status.idle":"2024-11-02T15:06:12.869466Z","shell.execute_reply.started":"2024-11-02T15:05:58.737433Z","shell.execute_reply":"2024-11-02T15:06:12.867816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_seq_red , y_test_seq)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T13:28:41.676693Z","iopub.execute_input":"2024-11-02T13:28:41.677128Z","iopub.status.idle":"2024-11-02T13:28:42.671453Z","shell.execute_reply.started":"2024-11-02T13:28:41.677090Z","shell.execute_reply":"2024-11-02T13:28:42.670490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = model.predict(test_seq_red)\n\nconv_preds = convert(preds , thresholds = [1 , 2 , 3])\nconv_test = convert(y_test_seq, thresholds = [1 , 2 , 3])\n\n\nquadratic_kappa(conv_test , conv_preds ,convert_=False )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-02T15:05:29.712030Z","iopub.execute_input":"2024-11-02T15:05:29.712815Z","iopub.status.idle":"2024-11-02T15:05:31.050486Z","shell.execute_reply.started":"2024-11-02T15:05:29.712769Z","shell.execute_reply":"2024-11-02T15:05:31.049367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}