{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-14T05:32:14.462012Z","iopub.execute_input":"2024-11-14T05:32:14.462805Z","iopub.status.idle":"2024-11-14T05:32:15.432223Z","shell.execute_reply.started":"2024-11-14T05:32:14.462757Z","shell.execute_reply":"2024-11-14T05:32:15.430916Z"},"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom lightgbm import LGBMClassifier\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:15.435021Z","iopub.execute_input":"2024-11-14T05:32:15.435401Z","iopub.status.idle":"2024-11-14T05:32:15.443342Z","shell.execute_reply.started":"2024-11-14T05:32:15.435357Z","shell.execute_reply":"2024-11-14T05:32:15.441817Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:15.445074Z","iopub.execute_input":"2024-11-14T05:32:15.446185Z","iopub.status.idle":"2024-11-14T05:32:15.520224Z","shell.execute_reply.started":"2024-11-14T05:32:15.446126Z","shell.execute_reply":"2024-11-14T05:32:15.518766Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:15.521343Z","iopub.execute_input":"2024-11-14T05:32:15.521761Z","iopub.status.idle":"2024-11-14T05:32:15.543384Z","shell.execute_reply.started":"2024-11-14T05:32:15.521716Z","shell.execute_reply":"2024-11-14T05:32:15.542089Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe(include = \"all\")","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:15.547192Z","iopub.execute_input":"2024-11-14T05:32:15.548039Z","iopub.status.idle":"2024-11-14T05:32:15.752656Z","shell.execute_reply.started":"2024-11-14T05:32:15.547982Z","shell.execute_reply":"2024-11-14T05:32:15.751042Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:15.754311Z","iopub.execute_input":"2024-11-14T05:32:15.754737Z","iopub.status.idle":"2024-11-14T05:32:15.788675Z","shell.execute_reply.started":"2024-11-14T05:32:15.754694Z","shell.execute_reply":"2024-11-14T05:32:15.787369Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:15.790039Z","iopub.execute_input":"2024-11-14T05:32:15.790391Z","iopub.status.idle":"2024-11-14T05:32:15.811802Z","shell.execute_reply.started":"2024-11-14T05:32:15.790355Z","shell.execute_reply":"2024-11-14T05:32:15.810282Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t = pd.read_parquet(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=7800a084/part-0.parquet\")\nt.head()\nprint(t.describe())\nprint(t.describe().values.reshape(-1))\ndel t\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:32:15.813671Z","iopub.execute_input":"2024-11-14T05:32:15.814077Z","iopub.status.idle":"2024-11-14T05:32:16.223810Z","shell.execute_reply.started":"2024-11-14T05:32:15.814036Z","shell.execute_reply":"2024-11-14T05:32:16.222399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reading a Parquet file, and returns descriptive statistics along with an identifier(id)\n\ndef process_file(filename: str , dirname: str)-> tuple:\n    df = pd.read_parquet(os.path.join(dirname,filename,'part-0.parquet'))\n    df.drop('step', axis = 1, inplace = True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:16.225354Z","iopub.execute_input":"2024-11-14T05:32:16.225744Z","iopub.status.idle":"2024-11-14T05:32:16.233712Z","shell.execute_reply.started":"2024-11-14T05:32:16.225705Z","shell.execute_reply":"2024-11-14T05:32:16.232001Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load time series data from directory and process each file using (process_file) \n# and return after aggregating results into dataframe \n\ndef load_time_series(dirname:str)->pd.DataFrame:\n    files = os.listdir(dirname) #gets all the files in dirname\n    \n    with ThreadPoolExecutor() as executor:\n        result_iter = executor.map(lambda filename: process_file(filename,dirname), files) # map process over each item in files list \n                                                                                           # and compute the lambda func which takes each filename of the files list as input\n        result_progress = tqdm(result_iter, total = len(files)) #to see the progress bar\n        result = list(result_progress) #format->tuple->(([1.2, 3.4, 5.6], 'file_1'))\n    statistics, indexes = zip(*result) # unwrap tuples and then aggregate all statistical results in 'statistics' & aggregate all indexes in 'indexes'\n    \n    df = pd.DataFrame(statistics, columns = [f\"stat_{i}\" for i in range(len(statistics[0]))])   # create DF with 'describe' statistics like mean,mas... as column\n                                                                                                # for each parquet files(996)\n    df['id'] = indexes #index (its matched with csv files)\n\n    return df\n    ","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:16.235123Z","iopub.execute_input":"2024-11-14T05:32:16.235540Z","iopub.status.idle":"2024-11-14T05:32:16.252172Z","shell.execute_reply.started":"2024-11-14T05:32:16.235468Z","shell.execute_reply":"2024-11-14T05:32:16.250970Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_time_series = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_time_series = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:32:16.253501Z","iopub.execute_input":"2024-11-14T05:32:16.253908Z","iopub.status.idle":"2024-11-14T05:34:37.194326Z","shell.execute_reply.started":"2024-11-14T05:32:16.253869Z","shell.execute_reply":"2024-11-14T05:34:37.192734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_time_series.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:34:37.196068Z","iopub.execute_input":"2024-11-14T05:34:37.196439Z","iopub.status.idle":"2024-11-14T05:34:37.231964Z","shell.execute_reply.started":"2024-11-14T05:34:37.196403Z","shell.execute_reply":"2024-11-14T05:34:37.230375Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"(time series features X describe()) = (12 X 8) = 96 ---> [stat_0,    stat_1,   \tstat_2,  \tstat_3,  \tstat_4,   ...   stat_93,  \tstat_94,  \tstat_95]","metadata":{}},{"cell_type":"code","source":"train_time_series.describe()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:34:37.233695Z","iopub.execute_input":"2024-11-14T05:34:37.234271Z","iopub.status.idle":"2024-11-14T05:34:37.428876Z","shell.execute_reply.started":"2024-11-14T05:34:37.234194Z","shell.execute_reply":"2024-11-14T05:34:37.427567Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_time_series.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.434120Z","iopub.execute_input":"2024-11-14T05:34:37.434574Z","iopub.status.idle":"2024-11-14T05:34:37.454385Z","shell.execute_reply.started":"2024-11-14T05:34:37.434530Z","shell.execute_reply":"2024-11-14T05:34:37.453058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_time_series.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T05:34:37.456203Z","iopub.execute_input":"2024-11-14T05:34:37.456721Z","iopub.status.idle":"2024-11-14T05:34:37.489099Z","shell.execute_reply.started":"2024-11-14T05:34:37.456666Z","shell.execute_reply":"2024-11-14T05:34:37.487583Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_time_series.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.490544Z","iopub.execute_input":"2024-11-14T05:34:37.490994Z","iopub.status.idle":"2024-11-14T05:34:37.510951Z","shell.execute_reply.started":"2024-11-14T05:34:37.490947Z","shell.execute_reply":"2024-11-14T05:34:37.509394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.merge(train,train_time_series,how = \"left\", on = \"id\")\ntest = pd.merge(test,test_time_series,how = \"left\", on = \"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.512650Z","iopub.execute_input":"2024-11-14T05:34:37.513088Z","iopub.status.idle":"2024-11-14T05:34:37.545542Z","shell.execute_reply.started":"2024-11-14T05:34:37.513045Z","shell.execute_reply":"2024-11-14T05:34:37.543175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.547291Z","iopub.execute_input":"2024-11-14T05:34:37.547881Z","iopub.status.idle":"2024-11-14T05:34:37.577792Z","shell.execute_reply.started":"2024-11-14T05:34:37.547820Z","shell.execute_reply":"2024-11-14T05:34:37.576103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.579837Z","iopub.execute_input":"2024-11-14T05:34:37.580802Z","iopub.status.idle":"2024-11-14T05:34:37.604971Z","shell.execute_reply.started":"2024-11-14T05:34:37.580732Z","shell.execute_reply":"2024-11-14T05:34:37.603348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.606496Z","iopub.execute_input":"2024-11-14T05:34:37.607143Z","iopub.status.idle":"2024-11-14T05:34:37.638833Z","shell.execute_reply.started":"2024-11-14T05:34:37.607079Z","shell.execute_reply":"2024-11-14T05:34:37.637503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.640724Z","iopub.execute_input":"2024-11-14T05:34:37.641333Z","iopub.status.idle":"2024-11-14T05:34:37.661410Z","shell.execute_reply.started":"2024-11-14T05:34:37.641284Z","shell.execute_reply":"2024-11-14T05:34:37.660036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#to exclude the columns in training set but not in test set as it ain't helping in prediction\ncolumns_train = []\nfor column in train.columns:\n    columns_train.append(column)\n    \ncolumns_test = []\nfor column in test.columns:\n    columns_test.append(column)\n    \ncommon_columns = [] \nfor column in columns_train:\n    if column in columns_test:\n        common_columns.append(column)\ncommon_columns.append('sii')\n\nexclude_columns = []\nfor column in columns_train:\n    if column not in common_columns:\n        exclude_columns.append(column)\n\ntrain = train[common_columns]\nprint(\"Columns that are excluded from training set: \")\nprint(exclude_columns)\ndel exclude_columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.662728Z","iopub.execute_input":"2024-11-14T05:34:37.663137Z","iopub.status.idle":"2024-11-14T05:34:37.679148Z","shell.execute_reply.started":"2024-11-14T05:34:37.663097Z","shell.execute_reply":"2024-11-14T05:34:37.677640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del columns_train\ndel columns_test\ndel common_columns\ndel test_time_series\ndel train_time_series","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.680825Z","iopub.execute_input":"2024-11-14T05:34:37.681333Z","iopub.status.idle":"2024-11-14T05:34:37.689675Z","shell.execute_reply.started":"2024-11-14T05:34:37.681281Z","shell.execute_reply":"2024-11-14T05:34:37.686968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.691617Z","iopub.execute_input":"2024-11-14T05:34:37.692195Z","iopub.status.idle":"2024-11-14T05:34:37.718082Z","shell.execute_reply.started":"2024-11-14T05:34:37.692149Z","shell.execute_reply":"2024-11-14T05:34:37.716515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop(columns=['id'])\ntest = test.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.720046Z","iopub.execute_input":"2024-11-14T05:34:37.721222Z","iopub.status.idle":"2024-11-14T05:34:37.734642Z","shell.execute_reply.started":"2024-11-14T05:34:37.721160Z","shell.execute_reply":"2024-11-14T05:34:37.733153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking no. of NAN values in each columns\ntotal_examples = len(train)\ntemp = pd.DataFrame(train.isnull().sum(), columns = ['NULL'])\ntemp['%NULL'] = (temp['NULL']*100)/total_examples\nprint(temp.sort_values('%NULL', ascending = False))\nprint(temp[temp.index.str.contains('sii', case=False, na=False)]) #explicitly checking null entries in target feature\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.736841Z","iopub.execute_input":"2024-11-14T05:34:37.737349Z","iopub.status.idle":"2024-11-14T05:34:37.759224Z","shell.execute_reply.started":"2024-11-14T05:34:37.737295Z","shell.execute_reply":"2024-11-14T05:34:37.757837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#counting no. of distinct types of sii values\nsns.countplot(x='sii', data = train)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.760885Z","iopub.execute_input":"2024-11-14T05:34:37.761416Z","iopub.status.idle":"2024-11-14T05:34:37.996399Z","shell.execute_reply.started":"2024-11-14T05:34:37.761359Z","shell.execute_reply":"2024-11-14T05:34:37.994916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.dropna(subset = 'sii')\ntrain = train.reset_index(drop = True)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:37.998313Z","iopub.execute_input":"2024-11-14T05:34:37.998884Z","iopub.status.idle":"2024-11-14T05:34:38.041360Z","shell.execute_reply.started":"2024-11-14T05:34:37.998824Z","shell.execute_reply":"2024-11-14T05:34:38.039943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:38.043930Z","iopub.execute_input":"2024-11-14T05:34:38.044328Z","iopub.status.idle":"2024-11-14T05:34:38.067926Z","shell.execute_reply.started":"2024-11-14T05:34:38.044290Z","shell.execute_reply":"2024-11-14T05:34:38.065904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Extracting numeric and categorical columns \n\n #train dataset\nnumeric_cols = list(train.select_dtypes(exclude = ['object']).columns.difference(['sii'])) #remove target column if target column is in the list\ncategorical_cols = list(train.select_dtypes(include =['object']).columns)\n\n#test dataset\ntest_numeric_cols = list(test.select_dtypes(exclude = ['object']).columns)\ntest_categorical_cols = list(test.select_dtypes(include =['object']).columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:38.069976Z","iopub.execute_input":"2024-11-14T05:34:38.071520Z","iopub.status.idle":"2024-11-14T05:34:38.083028Z","shell.execute_reply.started":"2024-11-14T05:34:38.071440Z","shell.execute_reply":"2024-11-14T05:34:38.081239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(numeric_cols))\nprint(numeric_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:38.084577Z","iopub.execute_input":"2024-11-14T05:34:38.085000Z","iopub.status.idle":"2024-11-14T05:34:38.097620Z","shell.execute_reply.started":"2024-11-14T05:34:38.084956Z","shell.execute_reply":"2024-11-14T05:34:38.096110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#boxplot of features(of csv)\n\nnum = len(numeric_cols) - 96 # 96-->time series features\nnum_per_batch = 10 \n# As the no. of features to plot is too large so, ploting in batchwise to make the plot reada\n\nfor i in range(0,num, num_per_batch):\n    end = i + num_per_batch\n    batch_cols = numeric_cols[i:end]\n\n    plt.figure(figsize = (14, 20))\n    for j,col in enumerate(batch_cols,1):\n        plt.subplot(5,2,j) # 5X2 subplots per batch\n        sns.boxplot(x='sii', y=col, data=train)\n        plt.title(f\"Boxplot of {col}\")\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:38.099032Z","iopub.execute_input":"2024-11-14T05:34:38.099449Z","iopub.status.idle":"2024-11-14T05:34:51.102233Z","shell.execute_reply.started":"2024-11-14T05:34:38.099403Z","shell.execute_reply":"2024-11-14T05:34:51.100787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#boxplot of features(of csv)\n\nnum = 96 # 96-->time series features\nnum_per_batch = 10 \n# As the no. of features to plot is too large so, ploting in batchwise to make the plot reada\n\nfor i in range(0,num, num_per_batch):\n    end = i + num_per_batch\n    batch_cols = numeric_cols[i:end]\n\n    plt.figure(figsize = (14, 20))\n    for j,col in enumerate(batch_cols,1):\n        plt.subplot(5,2,j) # 5X2 subplots per batch\n        sns.boxplot(x='sii', y=col, data=train)\n        plt.title(f\"Boxplot of {col}\")\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:34:51.103996Z","iopub.execute_input":"2024-11-14T05:34:51.104384Z","iopub.status.idle":"2024-11-14T05:35:17.771755Z","shell.execute_reply.started":"2024-11-14T05:34:51.104343Z","shell.execute_reply":"2024-11-14T05:35:17.770413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(f\"train{len(numeric_cols)}\")\nprint(numeric_cols)\nprint(f\"test{len(test_numeric_cols)}\")\nprint(test_numeric_cols)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:17.773344Z","iopub.execute_input":"2024-11-14T05:35:17.773938Z","iopub.status.idle":"2024-11-14T05:35:17.781518Z","shell.execute_reply.started":"2024-11-14T05:35:17.773707Z","shell.execute_reply":"2024-11-14T05:35:17.780137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#FILLING MISSING DATA WITH MEDIAN (NUMERICAL FEATUES)\n\n#train\nimputer = SimpleImputer(strategy='median')\ntrain[numeric_cols] = imputer.fit_transform(train[numeric_cols])\n\n#test\ntest[test_numeric_cols] = imputer.fit_transform(test[test_numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:17.783174Z","iopub.execute_input":"2024-11-14T05:35:17.783587Z","iopub.status.idle":"2024-11-14T05:35:17.890242Z","shell.execute_reply.started":"2024-11-14T05:35:17.783546Z","shell.execute_reply":"2024-11-14T05:35:17.888999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#CATEGORICAL FEATURES\n\nfor col in categorical_cols:\n    train[col] = train[col].fillna('missing')\n    train[col] = train[col].astype('category')\n    \n    test[col] = test[col].fillna('missing')\n    test[col] = test[col].astype('category') #uses less memory category dtype stores repeated values as int internally #makes operation easy(sorting,grouping) #LightGBM and XGBoost, work better with categorical features when the dtype is explicitly set to category\n\n#LABEL ENCODING\n\nle = LabelEncoder()\n\n#train\nfor col in categorical_cols:\n    train[col] = le.fit_transform(train[col]).astype(int)\n#test\nfor col in test_categorical_cols:\n     test[col] = le.transform(test[col]).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:17.892264Z","iopub.execute_input":"2024-11-14T05:35:17.892827Z","iopub.status.idle":"2024-11-14T05:35:17.942877Z","shell.execute_reply.started":"2024-11-14T05:35:17.892715Z","shell.execute_reply":"2024-11-14T05:35:17.941608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x = train.drop(['sii'], axis = 1)\ny = train['sii']\n\n# FEATURE SCALLING \n\nscaler = StandardScaler()\n    #train\nx[numeric_cols] = scaler.fit_transform(x[numeric_cols])\n    #test\ntest[numeric_cols] = scaler.transform(test[numeric_cols])\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:17.944225Z","iopub.execute_input":"2024-11-14T05:35:17.944657Z","iopub.status.idle":"2024-11-14T05:35:18.012294Z","shell.execute_reply.started":"2024-11-14T05:35:17.944613Z","shell.execute_reply":"2024-11-14T05:35:18.010714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Spliting training and validation set\nX_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:18.013835Z","iopub.execute_input":"2024-11-14T05:35:18.014221Z","iopub.status.idle":"2024-11-14T05:35:18.033465Z","shell.execute_reply.started":"2024-11-14T05:35:18.014178Z","shell.execute_reply":"2024-11-14T05:35:18.032084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nlgb_params = {\n    'lambda_l1': 1.84987772740316, \n    'lambda_l2': 2.74799424176787, \n    'num_leaves': 170, \n    'feature_fraction': 0.6557732374612563, \n    'bagging_fraction': 0.5511820667548543, \n    'bagging_freq': 3, \n    'min_child_samples': 94\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:18.035167Z","iopub.execute_input":"2024-11-14T05:35:18.035597Z","iopub.status.idle":"2024-11-14T05:35:18.042549Z","shell.execute_reply.started":"2024-11-14T05:35:18.035554Z","shell.execute_reply":"2024-11-14T05:35:18.040955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LGBMClassifier(**lgb_params, verbosity = -1)\nmodel.fit(X_train, y_train)\ny_train_pred = model.predict(X_train)\ny_val_pred = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:18.044739Z","iopub.execute_input":"2024-11-14T05:35:18.045146Z","iopub.status.idle":"2024-11-14T05:35:19.092993Z","shell.execute_reply.started":"2024-11-14T05:35:18.045105Z","shell.execute_reply":"2024-11-14T05:35:19.091814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\n\naccuracy = accuracy_score(y_test,y_val_pred)\nprint(f\"Validation Accuracy :{accuracy}\")\ncm = confusion_matrix(y_test,y_val_pred)\nprint(cm)\n\naccuracy = accuracy_score(y_train,y_train_pred)\nprint(f\"Train Accuracy :{accuracy}\")\ncm = confusion_matrix(y_train,y_train_pred)\nprint(cm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:19.099419Z","iopub.execute_input":"2024-11-14T05:35:19.100411Z","iopub.status.idle":"2024-11-14T05:35:19.118886Z","shell.execute_reply.started":"2024-11-14T05:35:19.100349Z","shell.execute_reply":"2024-11-14T05:35:19.117552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_actual = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")\ny_actual.head()\ny = y_actual['sii']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:19.120227Z","iopub.execute_input":"2024-11-14T05:35:19.120618Z","iopub.status.idle":"2024-11-14T05:35:19.133976Z","shell.execute_reply.started":"2024-11-14T05:35:19.120578Z","shell.execute_reply":"2024-11-14T05:35:19.132532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_predict = model.predict(test)\naccuracy = accuracy_score(y,final_predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:19.135141Z","iopub.execute_input":"2024-11-14T05:35:19.135571Z","iopub.status.idle":"2024-11-14T05:35:19.154186Z","shell.execute_reply.started":"2024-11-14T05:35:19.135526Z","shell.execute_reply":"2024-11-14T05:35:19.152369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output = pd.DataFrame({'id': y_actual['id'],'sii':final_predict})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T05:35:19.156007Z","iopub.execute_input":"2024-11-14T05:35:19.156666Z","iopub.status.idle":"2024-11-14T05:35:19.167040Z","shell.execute_reply.started":"2024-11-14T05:35:19.156460Z","shell.execute_reply":"2024-11-14T05:35:19.165805Z"}},"outputs":[],"execution_count":null}]}