{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-08T09:58:18.313883Z","iopub.execute_input":"2023-05-08T09:58:18.314305Z","iopub.status.idle":"2023-05-08T09:58:18.325731Z","shell.execute_reply.started":"2023-05-08T09:58:18.314264Z","shell.execute_reply":"2023-05-08T09:58:18.324596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-05-08T09:58:18.335706Z","iopub.execute_input":"2023-05-08T09:58:18.336127Z","iopub.status.idle":"2023-05-08T09:58:18.341105Z","shell.execute_reply.started":"2023-05-08T09:58:18.336087Z","shell.execute_reply":"2023-05-08T09:58:18.340140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndf = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T09:58:18.343082Z","iopub.execute_input":"2023-05-08T09:58:18.343822Z","iopub.status.idle":"2023-05-08T10:00:31.427556Z","shell.execute_reply.started":"2023-05-08T09:58:18.343773Z","shell.execute_reply":"2023-05-08T10:00:31.426136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:31.429642Z","iopub.execute_input":"2023-05-08T10:00:31.430057Z","iopub.status.idle":"2023-05-08T10:00:31.833082Z","shell.execute_reply.started":"2023-05-08T10:00:31.430022Z","shell.execute_reply":"2023-05-08T10:00:31.831979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:31.834685Z","iopub.execute_input":"2023-05-08T10:00:31.835436Z","iopub.status.idle":"2023-05-08T10:00:31.859213Z","shell.execute_reply.started":"2023-05-08T10:00:31.835391Z","shell.execute_reply":"2023-05-08T10:00:31.858340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:31.861661Z","iopub.execute_input":"2023-05-08T10:00:31.862256Z","iopub.status.idle":"2023-05-08T10:00:32.686626Z","shell.execute_reply.started":"2023-05-08T10:00:31.862206Z","shell.execute_reply":"2023-05-08T10:00:32.685676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:32.688064Z","iopub.execute_input":"2023-05-08T10:00:32.688654Z","iopub.status.idle":"2023-05-08T10:00:32.722421Z","shell.execute_reply.started":"2023-05-08T10:00:32.688617Z","shell.execute_reply":"2023-05-08T10:00:32.721360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Since the session_id is both a combination of both question number and session, we need to split it into two separate columns where first part represent the session before the '-' and second part represents the Question \nlabel_df['session'] = label_df.session_id.apply(lambda x: int(x.split('_')[0]))\nlabel_df['question'] = label_df.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\n\n#Let us understand this block of code:\n\n#With lambda x we are basically describing an anonymous function x which will split the session_id into two parts according to our preference\n#Next x.split('-')[0] signifies that we are splitting the code present on the left of '-' sign which is nothing but our session\n#Now, x.split('-')[-1][1:] shows that we are getting the second part of session_id","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:32.723919Z","iopub.execute_input":"2023-05-08T10:00:32.724533Z","iopub.status.idle":"2023-05-08T10:00:33.439548Z","shell.execute_reply.started":"2023-05-08T10:00:32.724494Z","shell.execute_reply":"2023-05-08T10:00:33.438435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:33.441032Z","iopub.execute_input":"2023-05-08T10:00:33.441664Z","iopub.status.idle":"2023-05-08T10:00:33.452311Z","shell.execute_reply.started":"2023-05-08T10:00:33.441624Z","shell.execute_reply":"2023-05-08T10:00:33.451315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:33.453892Z","iopub.execute_input":"2023-05-08T10:00:33.454520Z","iopub.status.idle":"2023-05-08T10:00:34.424867Z","shell.execute_reply.started":"2023-05-08T10:00:33.454482Z","shell.execute_reply":"2023-05-08T10:00:34.423558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(data=label_df, hue='correct')","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:00:34.426554Z","iopub.execute_input":"2023-05-08T10:00:34.427055Z","iopub.status.idle":"2023-05-08T10:01:39.122332Z","shell.execute_reply.started":"2023-05-08T10:00:34.427014Z","shell.execute_reply":"2023-05-08T10:01:39.121280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x=label_df['question'], hue=label_df['correct'], bins=20)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:01:39.126439Z","iopub.execute_input":"2023-05-08T10:01:39.126814Z","iopub.status.idle":"2023-05-08T10:01:39.645398Z","shell.execute_reply.started":"2023-05-08T10:01:39.126777Z","shell.execute_reply":"2023-05-08T10:01:39.644318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:01:39.646853Z","iopub.execute_input":"2023-05-08T10:01:39.647152Z","iopub.status.idle":"2023-05-08T10:01:39.680954Z","shell.execute_reply.started":"2023-05-08T10:01:39.647121Z","shell.execute_reply":"2023-05-08T10:01:39.679676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Next step is to check for categorical and numerical variables:\ncategorical_var = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nnumerical_var = ['elapsed_time','level','page','room_coor_x', 'room_coor_y','screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:01:39.682736Z","iopub.execute_input":"2023-05-08T10:01:39.683143Z","iopub.status.idle":"2023-05-08T10:01:39.689513Z","shell.execute_reply.started":"2023-05-08T10:01:39.683102Z","shell.execute_reply":"2023-05-08T10:01:39.688403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explaination for the code written below:\n* Defined a function named function and passed df_ as a parameter\n* iterating variable i in the categorical features defined above and then grouping all the features one by one on the basis of session_id and level_group (as mentioned in the data) and counting the number of unique values present in every feature\n* Next, we are concating or appending the new data we got with the array df_new already defined above \n* These same steps are followed for numerical features except that here we are calculate mean and standard deviation for every column and respectively concat both to the array df_new\n* After all the calculations, we finally need to add the df_new to our parameter df_ (column-wise) and fill all the null values with -1.\n* Then , finally create the data frame and set the index to session_id.","metadata":{}},{"cell_type":"code","source":"def function(df_):\n    df_new = []\n    for i in categorical_var:\n        data=df_.groupby(['session_id', 'level_group'])[i].agg('nunique')\n        df_new.append(data)\n    for i in numerical_var:\n        data=df_.groupby(['session_id', 'level_group'])[i].agg('mean')\n        df_new.append(data)\n    for i in numerical_var:\n        data=df_.groupby(['session_id', 'level_group'])[i].agg('std')\n        df_new.append(data)\n    df_=pd.concat(df_new, axis=1)\n    df_= df_.fillna(-1)\n    df_= df_.reset_index()\n    df_= df_.set_index('session_id')\n    return df_","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:01:39.690816Z","iopub.execute_input":"2023-05-08T10:01:39.691299Z","iopub.status.idle":"2023-05-08T10:01:39.703050Z","shell.execute_reply.started":"2023-05-08T10:01:39.691263Z","shell.execute_reply":"2023-05-08T10:01:39.701755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=function(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:01:39.704378Z","iopub.execute_input":"2023-05-08T10:01:39.705299Z","iopub.status.idle":"2023-05-08T10:02:19.257494Z","shell.execute_reply.started":"2023-05-08T10:01:39.705201Z","shell.execute_reply":"2023-05-08T10:02:19.256377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.259171Z","iopub.execute_input":"2023-05-08T10:02:19.260452Z","iopub.status.idle":"2023-05-08T10:02:19.266729Z","shell.execute_reply.started":"2023-05-08T10:02:19.260406Z","shell.execute_reply":"2023-05-08T10:02:19.265674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dropna()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.268311Z","iopub.execute_input":"2023-05-08T10:02:19.269575Z","iopub.status.idle":"2023-05-08T10:02:19.327740Z","shell.execute_reply.started":"2023-05-08T10:02:19.269521Z","shell.execute_reply":"2023-05-08T10:02:19.326552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.329092Z","iopub.execute_input":"2023-05-08T10:02:19.329711Z","iopub.status.idle":"2023-05-08T10:02:19.336661Z","shell.execute_reply.started":"2023-05-08T10:02:19.329666Z","shell.execute_reply":"2023-05-08T10:02:19.335363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature = [c for c in df.columns if c != 'level_group']\nlen(feature)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.338357Z","iopub.execute_input":"2023-05-08T10:02:19.338734Z","iopub.status.idle":"2023-05-08T10:02:19.353201Z","shell.execute_reply.started":"2023-05-08T10:02:19.338697Z","shell.execute_reply":"2023-05-08T10:02:19.350856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"users = df.index.unique()\nlen(users)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.354980Z","iopub.execute_input":"2023-05-08T10:02:19.355988Z","iopub.status.idle":"2023-05-08T10:02:19.365068Z","shell.execute_reply.started":"2023-05-08T10:02:19.355941Z","shell.execute_reply":"2023-05-08T10:02:19.363873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, train_test_split, KFold, GroupKFold\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score ","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.366853Z","iopub.execute_input":"2023-05-08T10:02:19.367228Z","iopub.status.idle":"2023-05-08T10:02:19.722372Z","shell.execute_reply.started":"2023-05-08T10:02:19.367194Z","shell.execute_reply":"2023-05-08T10:02:19.721199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def split_dataset(dataset, test_ratio=0.20):\n    user = df.index.unique()\n    split = int(len(user) * (1 - 0.20))\n    return dataset.loc[user[:split]], dataset.loc[user[split:]]\n\nx_train, x_test = split_dataset(df)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T11:19:14.689257Z","iopub.execute_input":"2023-05-08T11:19:14.689685Z","iopub.status.idle":"2023-05-08T11:19:14.935540Z","shell.execute_reply.started":"2023-05-08T11:19:14.689648Z","shell.execute_reply":"2023-05-08T11:19:14.934333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_score=[0.25, 0.5, 0.75, 0.1]\nn_estimators=[100, 200, 500, 900, 1500, 2000]\nmax_depth=[2,3,5,10,12,15]\nbooster=['gbtree', 'gblinear']\nlearning_rate=[0.01,0.05, 0.1, 0.2, 0.5]\nmin_child_weight=[1,2,3,4,5]\nhyperparemeter_grid={'n_estimators':n_estimators, 'max_depth':max_depth, 'learning_rate':learning_rate,\n                    'min_child_weight':min_child_weight, 'booster':booster, 'base_score':base_score }","metadata":{"execution":{"iopub.status.busy":"2023-05-08T11:19:24.690510Z","iopub.execute_input":"2023-05-08T11:19:24.691233Z","iopub.status.idle":"2023-05-08T11:19:24.697597Z","shell.execute_reply.started":"2023-05-08T11:19:24.691189Z","shell.execute_reply":"2023-05-08T11:19:24.696461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-08T11:19:27.015338Z","iopub.execute_input":"2023-05-08T11:19:27.016013Z","iopub.status.idle":"2023-05-08T11:19:27.022667Z","shell.execute_reply.started":"2023-05-08T11:19:27.015970Z","shell.execute_reply":"2023-05-08T11:19:27.021350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#You may also iterate thorugh all the questions and then check for the groups \n#for i in (1,19): (since there are in total 18 questions)\n#if(i<=3): group = '0-4'\n#elif(i<=13): group='5-12'\n#elif(i<=22): group='13-22'\n\n#Or you may also use map or replace function.\nnew_levels =[{ 'level_group':[0,0,0,0], 'question_id':[1,2,3,4]},{ 'level_group':[1,1,1,1,1,1,1,1], 'question_id':[5,6,7,8,9,10,11,12]},{'level_group':[2,2,2,2,2,2], 'question_id':[13,14,15,16,17,18]}]","metadata":{"execution":{"iopub.status.busy":"2023-05-08T10:02:19.734461Z","iopub.execute_input":"2023-05-08T10:02:19.735474Z","iopub.status.idle":"2023-05-08T10:02:19.746313Z","shell.execute_reply.started":"2023-05-08T10:02:19.735430Z","shell.execute_reply":"2023-05-08T10:02:19.745144Z"},"trusted":true},"execution_count":null,"outputs":[]}]}