{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-07T13:21:38.898639Z","iopub.execute_input":"2023-06-07T13:21:38.899034Z","iopub.status.idle":"2023-06-07T13:21:38.952158Z","shell.execute_reply.started":"2023-06-07T13:21:38.899003Z","shell.execute_reply":"2023-06-07T13:21:38.951249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reference\n\nhttps://www.kaggle.com/code/gusthema/student-performance-w-tensorflow-decision-forests\nI have understood Gusthema's code and did some changes and implemented in my way. The notebook was extremely useful in understanding the data in the form of sessions and levels. The way he feature engineered the columns was great.","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:21:38.954178Z","iopub.execute_input":"2023-06-07T13:21:38.954802Z","iopub.status.idle":"2023-06-07T13:21:39.451949Z","shell.execute_reply.started":"2023-06-07T13:21:38.954769Z","shell.execute_reply":"2023-06-07T13:21:39.451076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ntrain_data = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv',dtype=dtypes)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:21:39.453266Z","iopub.execute_input":"2023-06-07T13:21:39.453743Z","iopub.status.idle":"2023-06-07T13:23:50.818737Z","shell.execute_reply.started":"2023-06-07T13:21:39.453716Z","shell.execute_reply":"2023-06-07T13:23:50.817762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels['session'] = labels.session_id.apply(lambda x: int(x.split('_')[0]))\nlabels['q'] = labels.session_id.apply(lambda x: int(x.split('_')[-1][1:]))\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:50.819977Z","iopub.execute_input":"2023-06-07T13:23:50.820368Z","iopub.status.idle":"2023-06-07T13:23:51.893182Z","shell.execute_reply.started":"2023-06-07T13:23:50.820342Z","shell.execute_reply":"2023-06-07T13:23:51.891985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize = (3,3))\nplot_df = labels.correct.value_counts()\nplot_df.plot(kind = 'pie',autopct='%1.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:51.896076Z","iopub.execute_input":"2023-06-07T13:23:51.896522Z","iopub.status.idle":"2023-06-07T13:23:52.103751Z","shell.execute_reply.started":"2023-06-07T13:23:51.896492Z","shell.execute_reply":"2023-06-07T13:23:52.102913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"\\\"Correct\\\" column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    #print(n, str(n))\n    ax = plt.subplot(6, 3, n)\n\n    # filter df and plot ticker on the new subplot axis\n    plot_df = labels.loc[labels.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['b', 'r'])\n    \n    # chart formatting\n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:52.108052Z","iopub.execute_input":"2023-06-07T13:23:52.110246Z","iopub.status.idle":"2023-06-07T13:23:55.164372Z","shell.execute_reply.started":"2023-06-07T13:23:52.110201Z","shell.execute_reply":"2023-06-07T13:23:55.163488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Categoric and Numeric Segrigation","metadata":{}},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y', \n        'screen_coor_x', 'screen_coor_y', 'hover_duration']","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:55.165282Z","iopub.execute_input":"2023-06-07T13:23:55.165557Z","iopub.status.idle":"2023-06-07T13:23:55.175716Z","shell.execute_reply.started":"2023-06-07T13:23:55.165532Z","shell.execute_reply":"2023-06-07T13:23:55.174454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Categorical unique count","metadata":{}},{"cell_type":"code","source":"for i in CATEGORICAL:\n    print(i,\":{}\".format(train_data[i].nunique()))","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:55.177134Z","iopub.execute_input":"2023-06-07T13:23:55.177438Z","iopub.status.idle":"2023-06-07T13:23:55.874372Z","shell.execute_reply.started":"2023-06-07T13:23:55.177414Z","shell.execute_reply":"2023-06-07T13:23:55.873324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\ndef feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:55.875983Z","iopub.execute_input":"2023-06-07T13:23:55.876305Z","iopub.status.idle":"2023-06-07T13:23:55.884637Z","shell.execute_reply.started":"2023-06-07T13:23:55.876279Z","shell.execute_reply":"2023-06-07T13:23:55.883842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df = feature_engineer(train_data)\ndataset_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:23:55.886075Z","iopub.execute_input":"2023-06-07T13:23:55.886660Z","iopub.status.idle":"2023-06-07T13:24:36.490301Z","shell.execute_reply.started":"2023-06-07T13:23:55.886632Z","shell.execute_reply":"2023-06-07T13:24:36.489440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:24:36.491714Z","iopub.execute_input":"2023-06-07T13:24:36.492326Z","iopub.status.idle":"2023-06-07T13:24:36.498019Z","shell.execute_reply.started":"2023-06-07T13:24:36.492296Z","shell.execute_reply":"2023-06-07T13:24:36.497042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:24:36.499342Z","iopub.execute_input":"2023-06-07T13:24:36.499639Z","iopub.status.idle":"2023-06-07T13:24:36.652394Z","shell.execute_reply.started":"2023-06-07T13:24:36.499615Z","shell.execute_reply":"2023-06-07T13:24:36.651581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figure, axis = plt.subplots(3, 2, figsize=(10, 10))\n\nfor name, data in dataset_df.groupby('level_group'):\n    axis[0, 0].plot(range(1, len(data['room_coor_x_std'])+1), data['room_coor_x_std'], label=name)\n    axis[0, 1].plot(range(1, len(data['room_coor_y_std'])+1), data['room_coor_y_std'], label=name)\n    axis[1, 0].plot(range(1, len(data['screen_coor_x_std'])+1), data['screen_coor_x_std'], label=name)\n    axis[1, 1].plot(range(1, len(data['screen_coor_y_std'])+1), data['screen_coor_y_std'], label=name)\n    axis[2, 0].plot(range(1, len(data['hover_duration'])+1), data['hover_duration_std'], label=name)\n    axis[2, 1].plot(range(1, len(data['elapsed_time_std'])+1), data['elapsed_time_std'], label=name)\n    \n\naxis[0, 0].set_title('room_coor_x')\naxis[0, 1].set_title('room_coor_y')\naxis[1, 0].set_title('screen_coor_x')\naxis[1, 1].set_title('screen_coor_y')\naxis[2, 0].set_title('hover_duration')\naxis[2, 1].set_title('elapsed_time_std')\n\nfor i in range(3):\n    axis[i, 0].legend()\n    axis[i, 1].legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:24:36.653566Z","iopub.execute_input":"2023-06-07T13:24:36.654545Z","iopub.status.idle":"2023-06-07T13:24:39.754976Z","shell.execute_reply.started":"2023-06-07T13:24:36.654516Z","shell.execute_reply":"2023-06-07T13:24:39.754039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Split training and validation data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain, valid = train_test_split(dataset_df, test_size = 0.2)\nprint('{} examples in train, {} examples in validation'.format(len(train),len(valid)))","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:24:39.759529Z","iopub.execute_input":"2023-06-07T13:24:39.759866Z","iopub.status.idle":"2023-06-07T13:24:40.674656Z","shell.execute_reply.started":"2023-06-07T13:24:39.759838Z","shell.execute_reply":"2023-06-07T13:24:40.673585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model selection and implementation","metadata":{}},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nVALID_USER_LIST = valid.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:24:40.675818Z","iopub.execute_input":"2023-06-07T13:24:40.676131Z","iopub.status.idle":"2023-06-07T13:24:40.684967Z","shell.execute_reply.started":"2023-06-07T13:24:40.676105Z","shell.execute_reply":"2023-06-07T13:24:40.683876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nfrom sklearn import ensemble\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=18: grp = '13-22'\n    \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = train.loc[train.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid.loc[valid.level_group == grp]\n    valid_users = valid_df.index.values\n    \n    # Select the labels for the related q_no.\n    train_labels = labels.loc[labels.q==q_no].set_index('session').loc[train_users]\n    valid_labels = labels.loc[labels.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n#     train_df.loc[:,\"correct\"] = train_labels[\"correct\"]\n#     valid_df.loc[:,\"correct\"] = valid_labels[\"correct\"]\n\n    \n    train_ds = train_df.loc[:, train_df.columns != 'level_group']\n    valid_ds = valid_df.loc[:, valid_df.columns != 'level_group']\n    \n#     train_y = train_ds['correct']\n#     train_x = train_ds.drop('correct',axis = 1)\n#     valid_y = valid_ds['correct']\n#     valid_x = valid_ds.drop('correct',axis = 1)\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    # train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    # valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n\n    # We will now create the Gradient Boosted Trees Model with default settings. \n    # By default the model is set to train for a classification task.\n    gbtm = ensemble.GradientBoostingClassifier()\n    \n    # Train the model.\n    gbtm.fit(train_ds,train_labels['correct'])\n\n    # Store the model\n    models[f'{grp}_{q_no}'] = gbtm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    \n    accuracy = gbtm.score(valid_ds, valid_labels['correct'])\n    evaluation_dict[q_no] = accuracy         \n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = gbtm.predict(valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-07T13:24:40.686754Z","iopub.execute_input":"2023-06-07T13:24:40.687154Z","iopub.status.idle":"2023-06-07T13:28:22.291259Z","shell.execute_reply.started":"2023-06-07T13:24:40.687118Z","shell.execute_reply":"2023-06-07T13:28:22.289893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Output for Gradient Boosting algorithm","metadata":{}},{"cell_type":"code","source":"# for name, value in evaluation_dict.items():\n#   print(f\"question {name}: accuracy {value:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:39:58.500461Z","iopub.execute_input":"2023-06-07T13:39:58.501121Z","iopub.status.idle":"2023-06-07T13:39:58.508888Z","shell.execute_reply.started":"2023-06-07T13:39:58.501082Z","shell.execute_reply":"2023-06-07T13:39:58.507924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:22.302699Z","iopub.execute_input":"2023-06-07T13:28:22.303048Z","iopub.status.idle":"2023-06-07T13:28:22.344714Z","shell.execute_reply.started":"2023-06-07T13:28:22.303020Z","shell.execute_reply":"2023-06-07T13:28:22.343692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# features = gbtm.feature_importances_\n# features\n# for i in models:\n#     importance = gbtm.feature_importances_\n#     indices = importance.argsort()[-5:][::-1]\n#     features[indices] += 1\n# features/= 18\n# # print the variable importance\n# sum = features.sum()\n# for i,v in enumerate(features):\n#     print('Feature: %d, Percentage: %f%%' % (i,v/sum*100))","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:22.346239Z","iopub.execute_input":"2023-06-07T13:28:22.346557Z","iopub.status.idle":"2023-06-07T13:28:22.369252Z","shell.execute_reply.started":"2023-06-07T13:28:22.346531Z","shell.execute_reply":"2023-06-07T13:28:22.368180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:22.370752Z","iopub.execute_input":"2023-06-07T13:28:22.371249Z","iopub.status.idle":"2023-06-07T13:28:22.384990Z","shell.execute_reply.started":"2023-06-07T13:28:22.371219Z","shell.execute_reply":"2023-06-07T13:28:22.384013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe of required size:\n# (no: of users in validation set x no: of questions) initialized to zero values\n# to store true values of the label `correct`. \ntrue_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    # Get the true labels.\n    tmp = labels.loc[labels.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:22.386659Z","iopub.execute_input":"2023-06-07T13:28:22.387793Z","iopub.status.idle":"2023-06-07T13:28:22.528610Z","shell.execute_reply.started":"2023-06-07T13:28:22.387760Z","shell.execute_reply":"2023-06-07T13:28:22.527661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:22.532947Z","iopub.execute_input":"2023-06-07T13:28:22.533882Z","iopub.status.idle":"2023-06-07T13:28:22.574949Z","shell.execute_reply.started":"2023-06-07T13:28:22.533836Z","shell.execute_reply":"2023-06-07T13:28:22.574072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()\nsample_submission = sample_sub\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        gbtm = models[f'{grp}_{t}']\n        test_ds = test_df.loc[:, test_df.columns != 'level_group']\n        predictions = gbtm.predict(test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:22.579066Z","iopub.execute_input":"2023-06-07T13:28:22.579417Z","iopub.status.idle":"2023-06-07T13:28:23.086354Z","shell.execute_reply.started":"2023-06-07T13:28:22.579389Z","shell.execute_reply":"2023-06-07T13:28:23.084965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-07T13:28:23.087884Z","iopub.execute_input":"2023-06-07T13:28:23.088253Z","iopub.status.idle":"2023-06-07T13:28:24.253579Z","shell.execute_reply.started":"2023-06-07T13:28:23.088225Z","shell.execute_reply":"2023-06-07T13:28:24.251999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}