{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-26T10:39:12.707492Z","iopub.execute_input":"2023-02-26T10:39:12.708005Z","iopub.status.idle":"2023-02-26T10:39:12.721836Z","shell.execute_reply.started":"2023-02-26T10:39:12.707962Z","shell.execute_reply":"2023-02-26T10:39:12.720595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:12.724221Z","iopub.execute_input":"2023-02-26T10:39:12.724636Z","iopub.status.idle":"2023-02-26T10:39:12.733894Z","shell.execute_reply.started":"2023-02-26T10:39:12.724576Z","shell.execute_reply":"2023-02-26T10:39:12.732754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ndf_labels = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:12.735512Z","iopub.execute_input":"2023-02-26T10:39:12.735856Z","iopub.status.idle":"2023-02-26T10:39:57.452443Z","shell.execute_reply.started":"2023-02-26T10:39:12.735820Z","shell.execute_reply":"2023-02-26T10:39:57.451242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= df.iloc[:5000]","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.454557Z","iopub.execute_input":"2023-02-26T10:39:57.454918Z","iopub.status.idle":"2023-02-26T10:39:57.461215Z","shell.execute_reply.started":"2023-02-26T10:39:57.454885Z","shell.execute_reply":"2023-02-26T10:39:57.460059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sep(row):\n temp = row.split('_')\n return int(temp[0])","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.463064Z","iopub.execute_input":"2023-02-26T10:39:57.463870Z","iopub.status.idle":"2023-02-26T10:39:57.481187Z","shell.execute_reply.started":"2023-02-26T10:39:57.463824Z","shell.execute_reply":"2023-02-26T10:39:57.480263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels['new_session_id'] = df_labels['session_id'].apply(sep)\ndf_labels.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.482896Z","iopub.execute_input":"2023-02-26T10:39:57.483581Z","iopub.status.idle":"2023-02-26T10:39:57.723840Z","shell.execute_reply.started":"2023-02-26T10:39:57.483544Z","shell.execute_reply":"2023-02-26T10:39:57.722189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels = df_labels.drop(columns='session_id')\ndf_labels.rename(columns={'new_session_id':'session_id'},inplace=True)\ndf_labels.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.726326Z","iopub.execute_input":"2023-02-26T10:39:57.726871Z","iopub.status.idle":"2023-02-26T10:39:57.742387Z","shell.execute_reply.started":"2023-02-26T10:39:57.726829Z","shell.execute_reply":"2023-02-26T10:39:57.740898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.merge(df,df_labels,how='left', on=['session_id'])\ndf1.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.745030Z","iopub.execute_input":"2023-02-26T10:39:57.745548Z","iopub.status.idle":"2023-02-26T10:39:57.834315Z","shell.execute_reply.started":"2023-02-26T10:39:57.745497Z","shell.execute_reply":"2023-02-26T10:39:57.833201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1['elapsed_time_enc'] = df1['elapsed_time'].astype(\"category\").cat.codes","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.836135Z","iopub.execute_input":"2023-02-26T10:39:57.836478Z","iopub.status.idle":"2023-02-26T10:39:57.845899Z","shell.execute_reply.started":"2023-02-26T10:39:57.836446Z","shell.execute_reply":"2023-02-26T10:39:57.844390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1['level_group_enc'] = df1['level_group'].astype(\"category\").cat.codes","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.849409Z","iopub.execute_input":"2023-02-26T10:39:57.849764Z","iopub.status.idle":"2023-02-26T10:39:57.863801Z","shell.execute_reply.started":"2023-02-26T10:39:57.849731Z","shell.execute_reply":"2023-02-26T10:39:57.862537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx = df1[['elapsed_time_enc','level_group_enc']].to_numpy()\ny = df1['correct']\nx_train,x_test,y_train,y_test = train_test_split(x,y,test_size=0.1)\nx_train[:5,:]","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.865670Z","iopub.execute_input":"2023-02-26T10:39:57.866510Z","iopub.status.idle":"2023-02-26T10:39:57.896264Z","shell.execute_reply.started":"2023-02-26T10:39:57.866470Z","shell.execute_reply":"2023-02-26T10:39:57.895259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier()\nmodel.fit(x_train,y_train)\ny_predicted = model.predict(x_test)\nfrom sklearn.metrics import accuracy_score, confusion_matrix\naccuracy_score(y_predicted,y_test)*100","metadata":{"execution":{"iopub.status.busy":"2023-02-26T10:39:57.898075Z","iopub.execute_input":"2023-02-26T10:39:57.898875Z","iopub.status.idle":"2023-02-26T10:40:11.111895Z","shell.execute_reply.started":"2023-02-26T10:39:57.898835Z","shell.execute_reply":"2023-02-26T10:40:11.110620Z"},"trusted":true},"execution_count":null,"outputs":[]}]}