{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Mengimport Library","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_addons as tfa\nimport tensorflow_decision_forests as tfdf\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-23T13:13:59.270128Z","iopub.execute_input":"2023-06-23T13:13:59.270507Z","iopub.status.idle":"2023-06-23T13:14:09.270059Z","shell.execute_reply.started":"2023-06-23T13:13:59.270476Z","shell.execute_reply":"2023-06-23T13:14:09.269080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Memuat Dataset","metadata":{}},{"cell_type":"markdown","source":"Dalam rangka mengoptimalkan penggunaan memori saat membaca dataset dari file CSV menggunakan Pandas, terdapat beberapa strategi yang dapat dilakukan. Salah satunya adalah dengan melakukan pengurangan ukuran tipe data yang digunakan oleh kolom-kolom numerik.\n\nSecara default, Pandas akan secara otomatis mendeteksi tipe data kolom saat memuat dataset. Misalnya, kolom numerik akan diassign sebagai int64, kolom float akan diassign sebagai float64, dan kolom string akan diassign sebagai objek (object dtype).\n\n1. Membuat dictionary dtypes yang mendefinisikan tipe data kolom-kolom dataset. Setiap kolom diberikan tipe data yang sesuai menggunakan tipe data dari NumPy (seperti np.int32, np.uint8, np.float32) dan kategori (category) untuk kolom-kolom yang berisi label atau kategori.Membuat dictionary dtypes yang mendefinisikan tipe data kolom-kolom dataset. Setiap kolom diberikan tipe data yang sesuai menggunakan tipe data dari NumPy (seperti np.int32, np.uint8, np.float32) dan kategori (category) untuk kolom-kolom yang berisi label atau kategori.\n2. Menggunakan metode shape pada DataFrame untuk mencetak bentuk (shape) dataset yang telah dibaca. Hasilnya akan dicetak dalam format string dengan menggunakan fungsi format().Menggunakan metode shape pada DataFrame untuk mencetak bentuk (shape) dataset yang telah dibaca. Hasilnya akan dicetak dalam format string dengan menggunakan fungsi format().  ","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/predict-student-performance-from-game-play/discussion/384359\ndtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndataset_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(dataset_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:14:09.271966Z","iopub.execute_input":"2023-06-23T13:14:09.273088Z","iopub.status.idle":"2023-06-23T13:16:14.347537Z","shell.execute_reply.started":"2023-06-23T13:14:09.273053Z","shell.execute_reply":"2023-06-23T13:16:14.346719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:16:14.348864Z","iopub.execute_input":"2023-06-23T13:16:14.349360Z","iopub.status.idle":"2023-06-23T13:16:14.390542Z","shell.execute_reply.started":"2023-06-23T13:16:14.349332Z","shell.execute_reply":"2023-06-23T13:16:14.389767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data yang hilang perkolom\nmissingValue = dataset_df.isnull().sum()\n\n#jumlah total data yang hilang\ntotalCells = np.product(dataset_df.shape)\ntotalMissing = missingValue.sum()\npercentMissing = (totalMissing/totalCells)*100\n\nprint(percentMissing)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:16:14.391799Z","iopub.execute_input":"2023-06-23T13:16:14.392275Z","iopub.status.idle":"2023-06-23T13:16:14.978359Z","shell.execute_reply.started":"2023-06-23T13:16:14.392248Z","shell.execute_reply":"2023-06-23T13:16:14.977386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Memuat Label","metadata":{}},{"cell_type":"markdown","source":"1. Membuat kolom baru bernama 'session' dengan menggunakan metode .apply() pada kolom 'session_id'. Fungsi lambda digunakan untuk menerapkan fungsi anonim yang membagi 'session_id' berdasarkan karakter '_' dan mengambil elemen pertama (indeks 0) sebagai bilangan bulat. Hasilnya disimpan dalam kolom 'session'.\n2. Membuat kolom baru bernama 'q' dengan menggunakan metode .apply() pada kolom 'session_id'. Fungsi lambda digunakan untuk menerapkan fungsi anonim yang membagi 'session_id' berdasarkan karakter '_' dan mengambil elemen terakhir (indeks -1) setelah menghilangkan karakter pertama (indeks 1) sebagai bilangan bulat. Hasilnya disimpan dalam kolom 'q'.","metadata":{}},{"cell_type":"code","source":"label = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\nlabel['session'] = label.session_id.apply(lambda x: int(x.split('_')[0]) )\nlabel['q'] = label.session_id.apply(lambda x: int(x.split('_')[-1][1:]) )\nlabel.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:16:14.981115Z","iopub.execute_input":"2023-06-23T13:16:14.981678Z","iopub.status.idle":"2023-06-23T13:16:16.482831Z","shell.execute_reply.started":"2023-06-23T13:16:14.981647Z","shell.execute_reply":"2023-06-23T13:16:16.481689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Full train label shape is {}\".format(label.shape))","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:16:16.484157Z","iopub.execute_input":"2023-06-23T13:16:16.484552Z","iopub.status.idle":"2023-06-23T13:16:16.489700Z","shell.execute_reply.started":"2023-06-23T13:16:16.484523Z","shell.execute_reply":"2023-06-23T13:16:16.488688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Diagram Batang untuk Kolom Label Bernilai Benar","metadata":{}},{"cell_type":"markdown","source":"1. Mengatur ukuran figure menggunakan fungsi figure(figsize=(10, 20)) dengan lebar 10 inci dan tinggi 20 inci.\n2. Mengatur jarak antara subplot menggunakan fungsi subplots_adjust(hspace=0.5, wspace=0.5) dengan jarak vertikal (hspace) dan jarak horizontal (wspace) masing-masing 0.5.\n3. Menetapkan judul utama menggunakan fungsi suptitle() dengan teks \"\"Correct\" column values for each question\" dan fontsize 14. Parameter y=0.94 digunakan untuk mengatur posisi vertikal judul.\n4. Melakukan perulangan dari 1 hingga 18 menggunakan range(1, 19).\n5. Pada setiap iterasi, membuat subplot baru dengan menggunakan subplot(6, 3, n), dengan 6 baris, 3 kolom, dan nomor subplot sesuai dengan nilai n.\n6. Memfilter DataFrame label untuk mengambil baris yang memiliki nilai kolom 'q' sama dengan n menggunakan loc[label.q == n].\n7. Menghitung jumlah nilai 'correct' yang unik dan menggambarkannya dalam bentuk diagram batang menggunakan value_counts() dan plot() dengan jenis 'bar'. Warna batang ditentukan sebagai ['b', 'c'].\n8. Mengatur judul subplot dengan menggunakan set_title() dan menambahkan teks \"Question\" diikuti dengan nomor n dalam bentuk string.\n9. Mengatur label sumbu x pada subplot menjadi kosong menggunakan set_xlabel(\"\").\n\nDengan demikian, program tersebut menghasilkan subplot-grid berukuran 6x3 dengan diagram batang yang menunjukkan jumlah nilai 'correct' untuk setiap pertanyaan. Setiap subplot memiliki judul yang menunjukkan nomor pertanyaan yang sesuai.\n\n\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 20))\nplt.subplots_adjust(hspace=0.5, wspace=0.5)\nplt.suptitle(\"\\\"Correct\\\" column values for each question\", fontsize=14, y=0.94)\nfor n in range(1,19):\n    #print(n, str(n))\n    ax = plt.subplot(6, 3, n)\n\n    # filter df and plot ticker on the new subplot axis\n    plot_df = label.loc[label.q == n]\n    plot_df = plot_df.correct.value_counts()\n    plot_df.plot(ax=ax, kind=\"bar\", color=['b', 'r'])\n    \n    # chart formatting\n    ax.set_title(\"Question \" + str(n))\n    ax.set_xlabel(\"\")","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:16:16.491202Z","iopub.execute_input":"2023-06-23T13:16:16.491545Z","iopub.status.idle":"2023-06-23T13:16:19.597611Z","shell.execute_reply.started":"2023-06-23T13:16:16.491517Z","shell.execute_reply":"2023-06-23T13:16:19.596634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Menyiapkan Dataset","metadata":{}},{"cell_type":"markdown","source":"Fungsi feature_engineer melakukan beberapa operasi pemrosesan data pada DataFrame input dataset_df, seperti menghitung jumlah nilai unik dan statistik ringkasan dari fitur kategorikal dan numerikal berdasarkan kelompok 'session_id' dan 'level_group'.","metadata":{}},{"cell_type":"code","source":"CATEGORICAL = ['event_name', 'name','fqid', 'room_fqid', 'text_fqid']\nNUMERICAL = ['elapsed_time','level','page','room_coor_x', 'room_coor_y','screen_coor_x', 'screen_coor_y', 'hover_duration']\n\ndef feature_engineer(dataset_df):\n    dfs = []\n    for c in CATEGORICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('nunique')\n        tmp.name = tmp.name + '_nunique'\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('mean')\n        dfs.append(tmp)\n    for c in NUMERICAL:\n        tmp = dataset_df.groupby(['session_id','level_group'])[c].agg('std')\n        tmp.name = tmp.name + '_std'\n        dfs.append(tmp)\n    dataset_df = pd.concat(dfs,axis=1)\n    dataset_df = dataset_df.fillna(-1)\n    dataset_df = dataset_df.reset_index()\n    dataset_df = dataset_df.set_index('session_id')\n    return dataset_df\n\ndataset_df = feature_engineer(dataset_df)\nprint(\"Full prepared dataset shape is {}\".format(dataset_df.shape))\n\ndataset_df.head(5)\n# Reference: https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:16:19.598679Z","iopub.execute_input":"2023-06-23T13:16:19.598994Z","iopub.status.idle":"2023-06-23T13:17:00.612192Z","shell.execute_reply.started":"2023-06-23T13:16:19.598968Z","shell.execute_reply":"2023-06-23T13:17:00.611427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:17:00.613568Z","iopub.execute_input":"2023-06-23T13:17:00.614130Z","iopub.status.idle":"2023-06-23T13:17:00.755384Z","shell.execute_reply.started":"2023-06-23T13:17:00.614101Z","shell.execute_reply":"2023-06-23T13:17:00.754413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribusi Data Numerik","metadata":{}},{"cell_type":"code","source":"figure, axis = plt.subplots(3, 2, figsize=(10, 10))\n\nfor name, data in dataset_df.groupby('level_group'):\n    axis[0, 0].plot(range(1, len(data['room_coor_x_std'])+1), data['room_coor_x_std'], label=name)\n    axis[0, 1].plot(range(1, len(data['room_coor_y_std'])+1), data['room_coor_y_std'], label=name)\n    axis[1, 0].plot(range(1, len(data['screen_coor_x_std'])+1), data['screen_coor_x_std'], label=name)\n    axis[1, 1].plot(range(1, len(data['screen_coor_y_std'])+1), data['screen_coor_y_std'], label=name)\n    axis[2, 0].plot(range(1, len(data['hover_duration'])+1), data['hover_duration_std'], label=name)\n    axis[2, 1].plot(range(1, len(data['elapsed_time_std'])+1), data['elapsed_time_std'], label=name)\n    \n\naxis[0, 0].set_title('room_coor_x')\naxis[0, 1].set_title('room_coor_y')\naxis[1, 0].set_title('screen_coor_x')\naxis[1, 1].set_title('screen_coor_y')\naxis[2, 0].set_title('hover_duration')\naxis[2, 1].set_title('elapsed_time_std')\n\nfor i in range(3):\n    axis[i, 0].legend()\n    axis[i, 1].legend()\n\nplt.show()\n\ndef split_dataset(dataset, test_ratio=0.20):\n    USER_LIST = dataset.index.unique()\n    split = int(len(USER_LIST) * (1 - 0.20))\n    return dataset.loc[USER_LIST[:split]], dataset.loc[USER_LIST[split:]]\n\ntrain_x, valid_x = split_dataset(dataset_df)\nprint(\"{} examples in training, {} examples in testing.\".format(\n    len(train_x), len(valid_x)))","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:17:00.756654Z","iopub.execute_input":"2023-06-23T13:17:00.757080Z","iopub.status.idle":"2023-06-23T13:17:04.050957Z","shell.execute_reply.started":"2023-06-23T13:17:00.757051Z","shell.execute_reply":"2023-06-23T13:17:04.050028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Membuat Model","metadata":{}},{"cell_type":"code","source":"rfm = tfdf.keras.RandomForestModel(hyperparameter_template=\"benchmark_rank1\")","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:17:04.052479Z","iopub.execute_input":"2023-06-23T13:17:04.053062Z","iopub.status.idle":"2023-06-23T13:17:04.236055Z","shell.execute_reply.started":"2023-06-23T13:17:04.053029Z","shell.execute_reply":"2023-06-23T13:17:04.234974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"# Fetch the unique list of user sessions in the validation dataset. We assigned \n# `session_id` as the index of our feature engineered dataset. Hence fetching \n# the unique values in the index column will give us a list of users in the \n# validation set.\nVALID_USER_LIST = valid_x.index.unique()\n\n# Create a dataframe for storing the predictions of each question for all users\n# in the validation set.\n# For this, the required size of the data frame is: \n# (no: of users in validation set  x no of questions).\n# We will initialize all the predicted values in the data frame to zero.\n# The dataframe's index column is the user `session_id`s. \nprediction_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\n\n# Create an empty dictionary to store the models created for each question.\nmodels = {}\n\n# Create an empty dictionary to store the evaluation score for each question.\nevaluation_dict ={}\n\n# Iterate through questions 1 to 18 to train models for each question, evaluate\n# the trained model and store the predicted values.\nfor q_no in range(1,19):\n\n    # Select level group for the question based on the q_no.\n    if q_no<=3: grp = '0-4'\n    elif q_no<=13: grp = '5-12'\n    elif q_no<=22: grp = '13-22'\n    print(\"### q_no\", q_no, \"grp\", grp)\n    \n        \n    # Filter the rows in the datasets based on the selected level group. \n    train_df = train_x.loc[train_x.level_group == grp]\n    train_users = train_df.index.values\n    valid_df = valid_x.loc[valid_x.level_group == grp]\n    valid_users = valid_df.index.values\n\n    # Select the labels for the related q_no.\n    train_labels = label.loc[label.q==q_no].set_index('session').loc[train_users]\n    valid_labels = label.loc[label.q==q_no].set_index('session').loc[valid_users]\n\n    # Add the label to the filtered datasets.\n    train_df[\"correct\"] = train_labels[\"correct\"]\n    valid_df[\"correct\"] = valid_labels[\"correct\"]\n\n    # There's one more step required before we can train the model. \n    # We need to convert the datatset from Pandas format (pd.DataFrame)\n    # into TensorFlow Datasets format (tf.data.Dataset).\n    # TensorFlow Datasets is a high performance data loading library \n    # which is helpful when training neural networks with accelerators like GPUs and TPUs.\n    # We are omitting `level_group`, since it is not needed for training anymore.\n    train_ds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df.loc[:, train_df.columns != 'level_group'], label=\"correct\")\n    valid_ds = tfdf.keras.pd_dataframe_to_tf_dataset(valid_df.loc[:, valid_df.columns != 'level_group'], label=\"correct\")\n\n    # We will now create the Random Forest Model with default settings. \n    # By default the model is set to train for a classification task.\n    rfm = tfdf.keras.RandomForestModel(num_trees = 1000, max_depth = 8)\n    rfm.compile(metrics=[\"accuracy\"])\n\n    # Train the model.\n    rfm.fit(x=train_ds)\n\n    # Store the model\n    models[f'{grp}_{q_no}'] = rfm\n\n    # Evaluate the trained model on the validation dataset and store the \n    # evaluation accuracy in the `evaluation_dict`.\n    inspector = rfm.make_inspector()\n    inspector.evaluation()\n    evaluation = rfm.evaluate(x=valid_ds,return_dict=True)\n    evaluation_dict[q_no] = evaluation[\"accuracy\"]         \n\n    # Use the trained model to make predictions on the validation dataset and \n    # store the predicted values in the `prediction_df` dataframe.\n    predict = rfm.predict(x=valid_ds)\n    prediction_df.loc[valid_users, q_no-1] = predict.flatten()     ","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:17:04.237432Z","iopub.execute_input":"2023-06-23T13:17:04.237771Z","iopub.status.idle":"2023-06-23T13:28:26.531111Z","shell.execute_reply.started":"2023-06-23T13:17:04.237743Z","shell.execute_reply":"2023-06-23T13:28:26.530100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Periksa Akurasi Model","metadata":{}},{"cell_type":"code","source":"for name, value in evaluation_dict.items():\n  print(f\"question {name}: accuracy {value:.4f}\")\n\nprint(\"\\nAverage accuracy\", sum(evaluation_dict.values())/18)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:28:26.532973Z","iopub.execute_input":"2023-06-23T13:28:26.533584Z","iopub.status.idle":"2023-06-23T13:28:26.539227Z","shell.execute_reply.started":"2023-06-23T13:28:26.533554Z","shell.execute_reply":"2023-06-23T13:28:26.538497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualisasi","metadata":{}},{"cell_type":"code","source":"tfdf.model_plotter.plot_model_in_colab(models['0-4_1'], tree_idx=0, max_depth=3)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:28:26.543102Z","iopub.execute_input":"2023-06-23T13:28:26.543744Z","iopub.status.idle":"2023-06-23T13:28:26.570744Z","shell.execute_reply.started":"2023-06-23T13:28:26.543689Z","shell.execute_reply":"2023-06-23T13:28:26.569742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inspector = models['0-4_1'].make_inspector()\n\nprint(f\"Available variable importances:\")\nfor importance in inspector.variable_importances().keys():\n  print(\"\\t\", importance)\n\n# Each line is: (feature name, (index of the feature), importance score)\ninspector.variable_importances()[\"NUM_AS_ROOT\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:28:26.572142Z","iopub.execute_input":"2023-06-23T13:28:26.572464Z","iopub.status.idle":"2023-06-23T13:28:26.583459Z","shell.execute_reply.started":"2023-06-23T13:28:26.572437Z","shell.execute_reply":"2023-06-23T13:28:26.582723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Menentukan Threshold","metadata":{}},{"cell_type":"markdown","source":"Mencari nilai ambang (threshold) terbaik yang menghasilkan skor F1 (F1 score) tertinggi. Program ini menggunakan serangkaian nilai ambang yang berbeda dan menghitung skor F1 menggunakan metrik F1Score dari TensorFlow Addons (tfa.metrics.F1Score) dengan variasi nilai ambang tersebut.","metadata":{}},{"cell_type":"code","source":"# Create a dataframe of required size:\n# (no: of users in validation set x no: of questions) initialized to zero values\n# to store true values of the label `correct`. \ntrue_df = pd.DataFrame(data=np.zeros((len(VALID_USER_LIST),18)), index=VALID_USER_LIST)\nfor i in range(18):\n    # Get the true labels.\n    tmp = label.loc[label.q == i+1].set_index('session').loc[VALID_USER_LIST]\n    true_df[i] = tmp.correct.values\n\nmax_score = 0; best_threshold = 0\n\n# Loop through threshold values from 0.4 to 0.8 and select the threshold with \n# the highest `F1 score`.\nfor threshold in np.arange(0.4,0.8,0.01):\n    metric = tfa.metrics.F1Score(num_classes=2,average=\"macro\",threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1))>threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    if f1_score > max_score:\n        max_score = f1_score\n        best_threshold = threshold\n        \nprint(\"Best threshold \", best_threshold, \"\\tF1 score \", max_score)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:28:26.584719Z","iopub.execute_input":"2023-06-23T13:28:26.585195Z","iopub.status.idle":"2023-06-23T13:28:27.825638Z","shell.execute_reply.started":"2023-06-23T13:28:26.585167Z","shell.execute_reply":"2023-06-23T13:28:27.824834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresholds = np.arange(0.4, 0.8, 0.01)\nf1_scores = []\n\nfor threshold in thresholds:\n    metric = tfa.metrics.F1Score(num_classes=2, average=\"macro\", threshold=threshold)\n    y_true = tf.one_hot(true_df.values.reshape((-1)), depth=2)\n    y_pred = tf.one_hot((prediction_df.values.reshape((-1)) > threshold).astype('int'), depth=2)\n    metric.update_state(y_true, y_pred)\n    f1_score = metric.result().numpy()\n    f1_scores.append(f1_score)\n\n# Plot the threshold vs F1 score\nplt.plot(thresholds, f1_scores)\nplt.xlabel(\"Threshold\")\nplt.ylabel(\"F1 Score\")\nplt.title(\"Threshold vs F1 Score\")\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:28:27.826953Z","iopub.execute_input":"2023-06-23T13:28:27.827444Z","iopub.status.idle":"2023-06-23T13:28:28.789339Z","shell.execute_reply.started":"2023-06-23T13:28:27.827403Z","shell.execute_reply":"2023-06-23T13:28:28.788238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Reference\n# https://www.kaggle.com/code/philculliton/basic-submission-demo\n# https://www.kaggle.com/code/cdeotte/random-forest-baseline-0-664/notebook\n\n\nimport jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n    test_df = feature_engineer(test)\n    grp = test_df.level_group.values[0]\n    a,b = limits[grp]\n    for t in range(a,b):\n        rfm = models[f'{grp}_{t}']\n        test_ds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df.loc[:, test_df.columns != 'level_group'])\n        predictions = rfm.predict(test_ds)\n        mask = sample_submission.session_id.str.contains(f'q{t}')\n        n_predictions = (predictions > best_threshold).astype(int)\n        sample_submission.loc[mask,'correct'] = n_predictions.flatten()\n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-23T13:28:28.790538Z","iopub.execute_input":"2023-06-23T13:28:28.790864Z","iopub.status.idle":"2023-06-23T13:28:34.601474Z","shell.execute_reply.started":"2023-06-23T13:28:28.790837Z","shell.execute_reply":"2023-06-23T13:28:34.600615Z"},"trusted":true},"execution_count":null,"outputs":[]}]}