{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:10:52.704649Z","iopub.execute_input":"2023-04-05T11:10:52.706023Z","iopub.status.idle":"2023-04-05T11:10:52.714141Z","shell.execute_reply.started":"2023-04-05T11:10:52.705964Z","shell.execute_reply":"2023-04-05T11:10:52.712628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:10:54.414491Z","iopub.execute_input":"2023-04-05T11:10:54.414939Z","iopub.status.idle":"2023-04-05T11:13:07.234173Z","shell.execute_reply.started":"2023-04-05T11:10:54.414903Z","shell.execute_reply":"2023-04-05T11:13:07.232625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reduce Memory Usage\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:13:07.236834Z","iopub.execute_input":"2023-04-05T11:13:07.237436Z","iopub.status.idle":"2023-04-05T11:13:07.256332Z","shell.execute_reply.started":"2023-04-05T11:13:07.237393Z","shell.execute_reply":"2023-04-05T11:13:07.254806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = reduce_memory_usage(df)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:13:26.297628Z","iopub.execute_input":"2023-04-05T11:13:26.298195Z","iopub.status.idle":"2023-04-05T11:13:55.215335Z","shell.execute_reply.started":"2023-04-05T11:13:26.298149Z","shell.execute_reply":"2023-04-05T11:13:55.214004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train_df.copy()\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:13:55.217824Z","iopub.execute_input":"2023-04-05T11:13:55.218270Z","iopub.status.idle":"2023-04-05T11:13:55.544966Z","shell.execute_reply.started":"2023-04-05T11:13:55.218193Z","shell.execute_reply":"2023-04-05T11:13:55.543630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train = train.drop(['fullscreen', 'music', 'index', 'page', 'hq'], axis=1)\nnew_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:14:00.627183Z","iopub.execute_input":"2023-04-05T11:14:00.627653Z","iopub.status.idle":"2023-04-05T11:14:01.052263Z","shell.execute_reply.started":"2023-04-05T11:14:00.627615Z","shell.execute_reply":"2023-04-05T11:14:01.051323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = new_train.session_id\nX = new_train.drop('session_id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:14:02.438410Z","iopub.execute_input":"2023-04-05T11:14:02.438916Z","iopub.status.idle":"2023-04-05T11:14:02.829317Z","shell.execute_reply.started":"2023-04-05T11:14:02.438870Z","shell.execute_reply":"2023-04-05T11:14:02.828058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split the data\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=123)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:14:03.226439Z","iopub.execute_input":"2023-04-05T11:14:03.227406Z","iopub.status.idle":"2023-04-05T11:14:16.780818Z","shell.execute_reply.started":"2023-04-05T11:14:03.227359Z","shell.execute_reply":"2023-04-05T11:14:16.779578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_missing_vals(df):\n    \"\"\"This function is written to check and visualize missing values percentage in a dataframe\n    (df) : take a dataframe as a input\n    returns: returns and vizualize bar plot of missing values percentage\n    \"\"\"\n    missing_vals = df.isnull().sum()\n    missing_percentage = (missing_vals / len(df)) * 100\n    print(missing_percentage)\n    print('->>>>>>>>>>>>  Visualizing bar plot -<<<<<<<<<<<<<<<<<<')\n    plt.bar(missing_vals.index, missing_percentage)\n    plt.xlabel('Column Names')\n    plt.ylabel('Percentage')\n    plt.title(\"Percentage of Missing values \")\n    plt.xticks(rotation=90)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T10:58:54.195558Z","iopub.execute_input":"2023-04-05T10:58:54.195920Z","iopub.status.idle":"2023-04-05T10:58:54.204158Z","shell.execute_reply.started":"2023-04-05T10:58:54.195882Z","shell.execute_reply":"2023-04-05T10:58:54.202959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_missing_vals(df):\n    \"\"\"\n    This function is used to fill missing numeric values in a dataframe.\n    (df) : takes a dataframe with missing values as an input\n    returns : dataframe with filled missing values\n    \"\"\"\n    num_vals = df.select_dtypes(exclude=['category'])\n    col_list = num_vals.columns.to_list()\n    for col in col_list:\n        if df[col].isnull().sum()>0:\n            print(col)\n            median = df[col].median()\n            df[col + '_median'] = df[col].fillna(median)\n            df.drop(col, axis=1, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:14:16.782761Z","iopub.execute_input":"2023-04-05T11:14:16.783142Z","iopub.status.idle":"2023-04-05T11:14:16.791454Z","shell.execute_reply.started":"2023-04-05T11:14:16.783096Z","shell.execute_reply":"2023-04-05T11:14:16.789921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_categorical(df, columns):\n    \"\"\"\n    This function takes a dataframe and a list of categorical column names as input,\n    performs one-hot encoding on the categorical columns, drops the original columns,\n    and returns the modified dataframe.\n\n    Args:\n    df: pandas dataframe\n    columns: list of categorical column names to be encoded\n\n    Returns:\n    pandas dataframe: dataframe with encoded categorical columns and original columns dropped\n    \"\"\"\n    # make a copy of the input dataframe to avoid modifying the original dataframe\n    df_encoded = df.copy()\n    \n    # instantiate a OneHotEncoder object\n    encoder = OneHotEncoder(handle_unknown='ignore')\n    \n    # iterate over the categorical columns\n    for col in columns:\n        # perform one-hot encoding on the categorical column and get the encoded feature names\n        print(f\"Encoding column '{col}'...\")\n        encoded_features = encoder.fit_transform(df_encoded[[col]])\n        print(f\"Column encoded'{col}' \")\n        feature_names = encoder.get_feature_names([col])\n        \n        # create a dataframe of the encodecd features\n        encoded_df = pd.DataFrame(encoded_features.toarray(), columns=feature_names)\n        \n        # drop the original categorical column and concatenate the encoded dataframe\n        df_encoded = pd.concat([df_encoded, encoded_df], axis=1)\n        df_encoded.drop(col, axis=1, inplace=True)\n    \n    return df_encoded\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T15:37:28.188447Z","iopub.execute_input":"2023-04-05T15:37:28.188920Z","iopub.status.idle":"2023-04-05T15:37:28.227641Z","shell.execute_reply.started":"2023-04-05T15:37:28.188871Z","shell.execute_reply":"2023-04-05T15:37:28.226543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check missing values percentage of X_train\ncheck_missing_vals(X_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T09:46:10.072763Z","iopub.execute_input":"2023-04-05T09:46:10.073390Z","iopub.status.idle":"2023-04-05T09:46:11.084698Z","shell.execute_reply.started":"2023-04-05T09:46:10.073342Z","shell.execute_reply":"2023-04-05T09:46:11.083065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check missing values percentage of X_test\ncheck_missing_vals(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T09:45:48.326788Z","iopub.execute_input":"2023-04-05T09:45:48.327909Z","iopub.status.idle":"2023-04-05T09:45:48.876867Z","shell.execute_reply.started":"2023-04-05T09:45:48.327852Z","shell.execute_reply":"2023-04-05T09:45:48.875415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T08:42:15.329068Z","iopub.execute_input":"2023-04-05T08:42:15.329855Z","iopub.status.idle":"2023-04-05T08:42:33.278423Z","shell.execute_reply.started":"2023-04-05T08:42:15.329811Z","shell.execute_reply":"2023-04-05T08:42:33.277307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# impute missing values with their median into X_train and X_test \nfill_missing_vals(X_train)\nfill_missing_vals(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T10:59:04.574678Z","iopub.execute_input":"2023-04-05T10:59:04.575818Z","iopub.status.idle":"2023-04-05T10:59:15.109269Z","shell.execute_reply.started":"2023-04-05T10:59:04.575768Z","shell.execute_reply":"2023-04-05T10:59:15.108020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_categorical(df, columns, batch_size=2):\n    \"\"\"\n    This function takes a dataframe and a list of categorical column names as input,\n    performs one-hot encoding on the categorical columns in batches, and returns a list\n    of dataframes with encoded categorical columns.\n\n    Args:\n    df: pandas dataframe\n    columns: list of categorical column names to be encoded\n    batch_size: number of columns to encode in each batch (default=2)\n\n    Returns:\n    list of pandas dataframes: dataframes with encoded categorical columns\n    \"\"\"\n    # make a copy of the input dataframe to avoid modifying the original dataframe\n    df_copy = df.copy()\n    \n    # instantiate a OneHotEncoder object\n    encoder = OneHotEncoder(handle_unknown='ignore')\n    \n    # split columns into batches\n    batches = [columns[i:i+batch_size] for i in range(0, len(columns), batch_size)]\n    \n    # initialize empty list to store dataframes\n    encoded_dfs = []\n    \n    # iterate over the batches of columns\n    for batch in batches:\n        # encode the columns in the batch\n        encoded_features = encoder.fit_transform(df_copy[batch])\n        feature_names = encoder.get_feature_names(batch)\n        encoded_df = pd.DataFrame(encoded_features.toarray(), columns=feature_names)\n        \n        # drop the original categorical columns and add the encoded dataframe to the list\n        df_copy.drop(batch, axis=1, inplace=True)\n        encoded_dfs.append(encoded_df)\n    \n    # add the remaining non-categorical columns to the last dataframe\n    non_categorical_cols = [col for col in df_copy.columns if col not in columns]\n    encoded_dfs[-1] = pd.concat([encoded_dfs[-1], df_copy[non_categorical_cols]], axis=1)\n    \n    return encoded_dfs\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:15:12.830168Z","iopub.execute_input":"2023-04-05T11:15:12.830677Z","iopub.status.idle":"2023-04-05T11:15:12.841779Z","shell.execute_reply.started":"2023-04-05T11:15:12.830633Z","shell.execute_reply":"2023-04-05T11:15:12.840431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# excluding 'level_group' column\ncat_col = ['event_name', 'name','text', 'fqid', 'room_fqid', 'text_fqid']\n# encode categorical columns\nencode_categorical(X_train, cat_col)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T11:15:20.066361Z","iopub.execute_input":"2023-04-05T11:15:20.067229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-05T10:27:40.959566Z","iopub.execute_input":"2023-04-05T10:27:40.960157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}