{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This solution shows how sometimes a particularly simple model can achieve *acceptable* results while maintaining potentially a high level of explainability (unlike other more complex solutions).\n\n**OUTLINE:**\n\n* \"q\" feature extraction from train_labels.csv\n* lightGBM training (default hyperparameters)\n* Best F1 score threshold research\n* Submission","metadata":{}},{"cell_type":"markdown","source":"**What we have learnt:**\nMuch of the correctness of the students' response depends on the question itself, the difficulty of which is increasing as one progresses, and not on the way the gameplay is played. However, gameplay data should by no means be labelled as irrelevant.","metadata":{}},{"cell_type":"markdown","source":"**Import libraries + definition of functions**","metadata":{}},{"cell_type":"code","source":"# IMPORT KAGGLE API\nimport jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:23.220328Z","iopub.execute_input":"2023-06-20T20:34:23.220790Z","iopub.status.idle":"2023-06-20T20:34:23.282225Z","shell.execute_reply.started":"2023-06-20T20:34:23.220749Z","shell.execute_reply":"2023-06-20T20:34:23.280978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.metrics import f1_score\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:23.284426Z","iopub.execute_input":"2023-06-20T20:34:23.284795Z","iopub.status.idle":"2023-06-20T20:34:25.004562Z","shell.execute_reply.started":"2023-06-20T20:34:23.284767Z","shell.execute_reply":"2023-06-20T20:34:25.002790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndef data_preprocessing(dataframe):\n    dataframe.index = dataframe[\"session_id\"].apply(lambda x: x.split(\"_\")[0])\n    dataframe[\"q\"] = dataframe[\"session_id\"].apply(lambda x: int(x.split(\"_q\")[1]))\n    return dataframe\n\ndef split_dataset(dataset, test_ratio=0.20):\n    #We took inspiration from the baseline solution proposed by the challenge authors themselves.\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ndef best_threshold(valid_y, preds):\n  # Define a range of threshold values\n  thresholds = np.arange(0, 1, 0.01)\n\n  # Initialize variables to store the best F1 score and threshold\n  best_f1 = 0\n  best_threshold = 0\n\n  # Iterate over each threshold value and calculate F1 score\n  for threshold in thresholds:\n      # Convert probabilities to binary predictions based on the threshold\n      y_pred = [1 if prob >= threshold else 0 for prob in preds]\n      curr_f1 = f1_score(valid_y, y_pred, average = \"macro\")\n\n      # Check if current F1 score is better than the previous best\n      if curr_f1 > best_f1:\n          best_f1 = curr_f1\n          best_threshold = threshold\n\n  print(\"Best F1 score:\", best_f1)\n  print(\"Best threshold:\", best_threshold)\n\n  return best_f1, best_threshold","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:25.006157Z","iopub.execute_input":"2023-06-20T20:34:25.006568Z","iopub.status.idle":"2023-06-20T20:34:25.019465Z","shell.execute_reply.started":"2023-06-20T20:34:25.006521Z","shell.execute_reply":"2023-06-20T20:34:25.018398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data loading\nlabels = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\")\n\n#data preprocessing\nlabels = data_preprocessing(labels)\n\n#data splitting\ntrain, valid = split_dataset(labels)\n\nX_train = train[\"q\"].values.reshape(-1,1)\ny_train = train[\"correct\"].values\nX_test = valid[\"q\"].values.reshape(-1,1)\ny_test = valid[\"correct\"].values","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:25.022031Z","iopub.execute_input":"2023-06-20T20:34:25.022668Z","iopub.status.idle":"2023-06-20T20:34:27.058897Z","shell.execute_reply.started":"2023-06-20T20:34:25.022635Z","shell.execute_reply":"2023-06-20T20:34:27.057235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nclf = lgb.LGBMClassifier()\nclf.fit(X_train, y_train)\npreds = clf.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:27.060714Z","iopub.execute_input":"2023-06-20T20:34:27.061120Z","iopub.status.idle":"2023-06-20T20:34:30.005762Z","shell.execute_reply.started":"2023-06-20T20:34:27.061084Z","shell.execute_reply":"2023-06-20T20:34:30.004513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_f1, best_t = best_threshold(y_test,preds)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:30.007108Z","iopub.execute_input":"2023-06-20T20:34:30.007535Z","iopub.status.idle":"2023-06-20T20:34:42.436903Z","shell.execute_reply.started":"2023-06-20T20:34:30.007501Z","shell.execute_reply":"2023-06-20T20:34:42.435072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SUBMISSION**","metadata":{}},{"cell_type":"code","source":"counter = 0\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (test, sample_submission) in iter_test:\n    ## users make predictions here using the test data\n    #print(sample_submission)\n    question = sample_submission[\"session_id\"].apply(lambda x: str(x).split(\"_q\")[1])\n    #print(question)\n    question = question.astype(int)\n    #print(model.predict(question))\n    sample_submission['correct'] = (clf.predict_proba(question.values.reshape(-1, 1))[:,1] > best_t)*1\n    #counter += 1\n    env.predict(sample_submission)\n    \n    \n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T20:34:42.438807Z","iopub.execute_input":"2023-06-20T20:34:42.439332Z","iopub.status.idle":"2023-06-20T20:34:42.545583Z","shell.execute_reply.started":"2023-06-20T20:34:42.439287Z","shell.execute_reply":"2023-06-20T20:34:42.544220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head()","metadata":{"papermill":{"duration":0.027432,"end_time":"2023-02-07T01:02:45.541022","exception":false,"start_time":"2023-02-07T01:02:45.51359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-20T20:34:42.548951Z","iopub.execute_input":"2023-06-20T20:34:42.549455Z","iopub.status.idle":"2023-06-20T20:34:42.827099Z","shell.execute_reply.started":"2023-06-20T20:34:42.549407Z","shell.execute_reply":"2023-06-20T20:34:42.825865Z"},"trusted":true},"execution_count":null,"outputs":[]}]}