{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:16.651198Z","iopub.execute_input":"2023-02-14T18:20:16.651598Z","iopub.status.idle":"2023-02-14T18:20:16.655949Z","shell.execute_reply.started":"2023-02-14T18:20:16.651567Z","shell.execute_reply":"2023-02-14T18:20:16.655045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install chart_studio\n","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:16.661391Z","iopub.execute_input":"2023-02-14T18:20:16.661765Z","iopub.status.idle":"2023-02-14T18:20:27.021502Z","shell.execute_reply.started":"2023-02-14T18:20:16.661722Z","shell.execute_reply":"2023-02-14T18:20:27.020503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport chart_studio.plotly as py\nimport cufflinks as cf\nimport seaborn as sns\nimport plotly.express as px\n%matplotlib inline\n\nimport plotly.io as pio\npio.renderers.default = 'colab'\n\n# Make Plotly work in your Jupyter Notebook\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\ninit_notebook_mode(connected=True)\n# Use Plotly locally\ncf.go_offline()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.023280Z","iopub.execute_input":"2023-02-14T18:20:27.023596Z","iopub.status.idle":"2023-02-14T18:20:27.038103Z","shell.execute_reply.started":"2023-02-14T18:20:27.023564Z","shell.execute_reply":"2023-02-14T18:20:27.037350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dtypes={'session_id':'category', \n# 'elapsed_time':np.int32,\n#     'event_name':'category',\n#     'name':'category',\n#     'level':np.uint8,\n#     'page':'category',\n#     'room_coor_x':np.float32,\n#     'room_coor_y':np.float32,\n#     'screen_coor_x':np.float32,\n#     'screen_coor_y':np.float32,\n#     'hover_duration':np.float32,\n#      'text':'category',\n#      'fqid':'category',\n#      'room_fqid':'category',\n#      'text_fqid':'category',\n#      'level_group':'category'}","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.038945Z","iopub.execute_input":"2023-02-14T18:20:27.039235Z","iopub.status.idle":"2023-02-14T18:20:27.052013Z","shell.execute_reply.started":"2023-02-14T18:20:27.039210Z","shell.execute_reply":"2023-02-14T18:20:27.051260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",dtype=dtypes,\n#                     usecols=lambda x: x not in ['fullscreen', 'hq', 'music'])\ntrain = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",usecols=lambda x: x not in ['fullscreen', 'hq', 'music'])\ntrain.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.053723Z","iopub.execute_input":"2023-02-14T18:20:27.054001Z","iopub.status.idle":"2023-02-14T18:20:27.333598Z","shell.execute_reply.started":"2023-02-14T18:20:27.053975Z","shell.execute_reply":"2023-02-14T18:20:27.332651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.335306Z","iopub.execute_input":"2023-02-14T18:20:27.335607Z","iopub.status.idle":"2023-02-14T18:20:27.360484Z","shell.execute_reply.started":"2023-02-14T18:20:27.335580Z","shell.execute_reply":"2023-02-14T18:20:27.359616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.session_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.361834Z","iopub.execute_input":"2023-02-14T18:20:27.362231Z","iopub.status.idle":"2023-02-14T18:20:27.368294Z","shell.execute_reply.started":"2023-02-14T18:20:27.362202Z","shell.execute_reply":"2023-02-14T18:20:27.367497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.369621Z","iopub.execute_input":"2023-02-14T18:20:27.369936Z","iopub.status.idle":"2023-02-14T18:20:27.379967Z","shell.execute_reply.started":"2023-02-14T18:20:27.369901Z","shell.execute_reply":"2023-02-14T18:20:27.379116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data type of each column-  info\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.383549Z","iopub.execute_input":"2023-02-14T18:20:27.383903Z","iopub.status.idle":"2023-02-14T18:20:27.424173Z","shell.execute_reply.started":"2023-02-14T18:20:27.383856Z","shell.execute_reply":"2023-02-14T18:20:27.423462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_data = train.nunique()\n\ndf_count = pd.DataFrame(unique_data).reset_index()\ndf_count.columns = ['train_column', 'count']\ndf_count.sort_values(by = 'count',ascending = False,inplace = True)\ndf_count","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.425143Z","iopub.execute_input":"2023-02-14T18:20:27.425811Z","iopub.status.idle":"2023-02-14T18:20:27.503842Z","shell.execute_reply.started":"2023-02-14T18:20:27.425783Z","shell.execute_reply":"2023-02-14T18:20:27.503102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing values on each columns\n\nmissing_val = train.isna().sum()\nmissing_val.sort_values(ascending = False,inplace = True)\nmissing_val","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.507190Z","iopub.execute_input":"2023-02-14T18:20:27.507500Z","iopub.status.idle":"2023-02-14T18:20:27.546506Z","shell.execute_reply.started":"2023-02-14T18:20:27.507472Z","shell.execute_reply":"2023-02-14T18:20:27.545551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events = train.event_name.value_counts()\nevents.sort_values(ascending = False,inplace = True)\nevents","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.547957Z","iopub.execute_input":"2023-02-14T18:20:27.548238Z","iopub.status.idle":"2023-02-14T18:20:27.561531Z","shell.execute_reply.started":"2023-02-14T18:20:27.548212Z","shell.execute_reply":"2023-02-14T18:20:27.560535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.name.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.562793Z","iopub.execute_input":"2023-02-14T18:20:27.563181Z","iopub.status.idle":"2023-02-14T18:20:27.580947Z","shell.execute_reply.started":"2023-02-14T18:20:27.563152Z","shell.execute_reply":"2023-02-14T18:20:27.579989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"level_groups = train.level_group.value_counts()\nlevel_groups.sort_values(ascending = False,inplace = True)\nlevel_groups","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.582406Z","iopub.execute_input":"2023-02-14T18:20:27.582720Z","iopub.status.idle":"2023-02-14T18:20:27.595963Z","shell.execute_reply.started":"2023-02-14T18:20:27.582693Z","shell.execute_reply":"2023-02-14T18:20:27.594922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"levels = train.level.value_counts()\nlevels.sort_values(ascending = False,inplace = True)\nlevels","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.597564Z","iopub.execute_input":"2023-02-14T18:20:27.597826Z","iopub.status.idle":"2023-02-14T18:20:27.606553Z","shell.execute_reply.started":"2023-02-14T18:20:27.597802Z","shell.execute_reply":"2023-02-14T18:20:27.605605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def enable_plotly_in_cell():\n  import IPython\n  from plotly.offline import init_notebook_mode\n  display(IPython.core.display.HTML('''<script src=\"/static/components/requirejs/require.js\"></script>'''))\n  init_notebook_mode(connected=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.607979Z","iopub.execute_input":"2023-02-14T18:20:27.608249Z","iopub.status.idle":"2023-02-14T18:20:27.616663Z","shell.execute_reply.started":"2023-02-14T18:20:27.608224Z","shell.execute_reply":"2023-02-14T18:20:27.615830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar with plotly.bar() \nfig = px.bar(df_count, y='count', x='train_column', text='count', color='train_column',width = 1200,height = 600,title= \"Number of Unique values on each columns\")\nfig\n\n# Put bar total value above bars with 2 values of precision\nfig.update_traces(texttemplate='%{text:.2s}', textposition='outside')\n\n# Set fontsize and uniformtext_mode='hide' says to hide the text if it won't fit\nfig.update_layout(uniformtext_minsize=8)\n\n# Rotate labels 45 degrees\nfig.update_layout(xaxis_tickangle=-45)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.617692Z","iopub.execute_input":"2023-02-14T18:20:27.618227Z","iopub.status.idle":"2023-02-14T18:20:27.759588Z","shell.execute_reply.started":"2023-02-14T18:20:27.618200Z","shell.execute_reply":"2023-02-14T18:20:27.758772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar with plotly.bar() \nfig = px.bar(missing_val, y=missing_val.values, x=missing_val.index, text=missing_val.values, \n             color=missing_val.index,width = 1200,height = 600,title= \"Missing Values on Each Columns\",\n             labels=dict(index= \"Column Name\", y =\"Number of missing Values\"))\nfig\n\n# Put bar total value above bars with 2 values of precision\nfig.update_traces(texttemplate='%{text:.2s}', textposition='outside')\n\n# Set fontsize and uniformtext_mode='hide' says to hide the text if it won't fit\nfig.update_layout(uniformtext_minsize=8)\n\n# Rotate labels 45 degrees\nfig.update_layout(xaxis_tickangle=-45)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.760591Z","iopub.execute_input":"2023-02-14T18:20:27.761398Z","iopub.status.idle":"2023-02-14T18:20:27.893771Z","shell.execute_reply.started":"2023-02-14T18:20:27.761368Z","shell.execute_reply":"2023-02-14T18:20:27.892970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar with plotly.bar() \nfig = px.bar(events, y=events.values, x=events.index, text=events.values, color=events.index,width = 1200,height = 600,\n             title= \"Number of Unique Events.\",labels=dict(index= \"Events\", y =\"Number of Unique Events.\"))\nfig\n\n# Put bar total value above bars with 2 values of precision\nfig.update_traces(texttemplate='%{text:.2s}', textposition='outside')\n\n# Set fontsize and uniformtext_mode='hide' says to hide the text if it won't fit\nfig.update_layout(uniformtext_minsize=8)\n\n# Rotate labels 45 degrees\nfig.update_layout(xaxis_tickangle=-45)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:27.894690Z","iopub.execute_input":"2023-02-14T18:20:27.895524Z","iopub.status.idle":"2023-02-14T18:20:28.002437Z","shell.execute_reply.started":"2023-02-14T18:20:27.895495Z","shell.execute_reply":"2023-02-14T18:20:28.001607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar with plotly.bar() \nfig = px.bar(level_groups, y=level_groups.values, x=level_groups.index, text=level_groups.values, color=level_groups.index,width = 600,height = 600,\n             title= \"Number of Level Groups\",labels=dict(index= \"Level Groups\", y =\"Number of level groups.\"))\nfig\n\n# Put bar total value above bars with 2 values of precision\nfig.update_traces(texttemplate='%{text:.2s}', textposition='outside')\n\n# Set fontsize and uniformtext_mode='hide' says to hide the text if it won't fit\nfig.update_layout(uniformtext_minsize=8)\n\n# Rotate labels 45 degrees\nfig.update_layout(xaxis_tickangle=-45)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.003568Z","iopub.execute_input":"2023-02-14T18:20:28.004505Z","iopub.status.idle":"2023-02-14T18:20:28.077983Z","shell.execute_reply.started":"2023-02-14T18:20:28.004470Z","shell.execute_reply":"2023-02-14T18:20:28.077276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar with plotly.bar() \nfig = px.bar(levels, y=levels.values, x=levels.index, text=levels.values, color=levels.index,width = 1200,height = 600,\n             title= \"Number of Levels.\",labels=dict(index= \"Levels\", y =\"Number of Unique levels.\"))\nfig.update_layout( xaxis = dict( tickmode = 'linear', tick0 = 0, dtick = 1))\nfig\n\n# Put bar total value above bars with 2 values of precision\nfig.update_traces(texttemplate='%{text:.2s}', textposition='outside')\n\n# Set fontsize and uniformtext_mode='hide' says to hide the text if it won't fit\nfig.update_layout(uniformtext_minsize=8)\n\n# Rotate labels 45 degrees\nfig.update_layout(xaxis_tickangle=0)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.079115Z","iopub.execute_input":"2023-02-14T18:20:28.079583Z","iopub.status.idle":"2023-02-14T18:20:28.147477Z","shell.execute_reply.started":"2023-02-14T18:20:28.079556Z","shell.execute_reply":"2023-02-14T18:20:28.146737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676\nCATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click','person_click','cutscene_click','object_click',\n          'map_hover','notification_click','map_click','observation_click',\n          'checkpoint']","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.148529Z","iopub.execute_input":"2023-02-14T18:20:28.148982Z","iopub.status.idle":"2023-02-14T18:20:28.154131Z","shell.execute_reply.started":"2023-02-14T18:20:28.148953Z","shell.execute_reply":"2023-02-14T18:20:28.153118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676\n\ndef feature_engineer(train):\n    \n    dfs = []\n    for c in CATS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('mean')\n        tmp.name = tmp.name + '_mean'\n        dfs.append(tmp)\n    for c in NUMS:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    for c in EVENTS: \n        train[c] = (train.event_name == c).astype('int8')\n    for c in EVENTS + ['elapsed_time']:\n        tmp = train.groupby(['session_id','level_group'])[c].agg('sum')\n        tmp.name = tmp.name + '_sum'\n        dfs.append(tmp)\n    train = train.drop(EVENTS,axis=1)\n        \n    df = pd.concat(dfs,axis=1)\n    df = df.fillna(-1)\n    df = df.reset_index()\n    df = df.set_index('session_id')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.155447Z","iopub.execute_input":"2023-02-14T18:20:28.155701Z","iopub.status.idle":"2023-02-14T18:20:28.168688Z","shell.execute_reply.started":"2023-02-14T18:20:28.155677Z","shell.execute_reply":"2023-02-14T18:20:28.167952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = feature_engineer(train)\nprint( df.shape )\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.170098Z","iopub.execute_input":"2023-02-14T18:20:28.170437Z","iopub.status.idle":"2023-02-14T18:20:28.697845Z","shell.execute_reply.started":"2023-02-14T18:20:28.170399Z","shell.execute_reply":"2023-02-14T18:20:28.697073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.698821Z","iopub.execute_input":"2023-02-14T18:20:28.699564Z","iopub.status.idle":"2023-02-14T18:20:28.712094Z","shell.execute_reply.started":"2023-02-14T18:20:28.699535Z","shell.execute_reply":"2023-02-14T18:20:28.711378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:20:28.712997Z","iopub.execute_input":"2023-02-14T18:20:28.713767Z","iopub.status.idle":"2023-02-14T18:20:28.745728Z","shell.execute_reply.started":"2023-02-14T18:20:28.713739Z","shell.execute_reply":"2023-02-14T18:20:28.744975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.index.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T18:21:27.217335Z","iopub.execute_input":"2023-02-14T18:21:27.217703Z","iopub.status.idle":"2023-02-14T18:21:27.224024Z","shell.execute_reply.started":"2023-02-14T18:21:27.217675Z","shell.execute_reply":"2023-02-14T18:21:27.223026Z"},"trusted":true},"execution_count":null,"outputs":[]}]}