{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#libraries\nimport numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nsns.set()\n#preprocess\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MinMaxScaler\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import LabelEncoder\n#models\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import VotingClassifier\nimport xgboost as xgb\nfrom sklearn.tree import DecisionTreeClassifier\n#check\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import precision_score\n#save\nimport pickle as pk\n# file\nimport os\nprint (os.listdir(\"../input\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtypes={\n    'elapsed_time':np.int32,\n    'event_name':'category',\n    'name':'category',\n    'level':np.uint8,\n    'room_coor_x':np.float32,\n    'room_coor_y':np.float32,\n    'screen_coor_x':np.float32,\n    'screen_coor_y':np.float32,\n    'hover_duration':np.float32,\n    'text':'category',\n    'fqid':'category',\n    'room_fqid':'category',\n    'text_fqid':'category',\n    'fullscreen':'category',\n    'hq':'category',\n    'music':'category',\n    'level_group':'category'}\n\ndata1 = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', dtype=dtypes)\nprint(\"Full train dataset shape is {}\".format(data1.shape))\n#-----------------------------------------------------part one\n","metadata":{"execution":{"iopub.status.busy":"2023-06-29T15:46:17.181297Z","iopub.execute_input":"2023-06-29T15:46:17.183944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#-----------------------------------------------------part one\n# read data\ndata2 = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv')\ndata2.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data2.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1.columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data2.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data2.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data2 = data2.fillna(0,axis=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1 = data1.fillna(0,axis=0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlacao_notas = data1.corr()\n\nplt.figure(figsize=(10, 6))\nsns.heatmap(correlacao_notas, annot=True, cmap=\"BrBG\", vmin=-1, vmax=1)\nplt.xticks(rotation=45)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train = train.select_dtypes(include = ['float64', 'int64'])\nTrain","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Test = test.select_dtypes(include = ['float64', 'int64'])\nTest","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlacao_notas = Train.corr()\n\nplt.figure(figsize=(200, 160))\nsns.heatmap(correlacao_notas, annot=True, cmap=\"BrBG\", vmin=-1, vmax=1)\nplt.xticks(rotation=45)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = Train.drop(columns=['session_id','session_level'])\ny= Train['session_level']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, cross_validate, train_test_split\nfrom sklearn.metrics import make_scorer, r2_score, mean_absolute_error\nfrom sklearn.linear_model import LinearRegression, Ridge\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.linear_model import LinearRegression, Ridge\nlr = LinearRegression()\nlr.fit(X_train, y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr.score(X_test, y_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mae = make_scorer(mean_absolute_error)\nr2 = make_scorer(r2_score)\n\ncvs = cross_validate(estimator=LinearRegression(), X=X_train, y=y_train, cv=10, verbose=10, scoring={'mae': mae, 'r2':r2})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The mean of the result is %.3f\" % (cvs['test_r2'].mean()))\nprint(\"The standard desviation error is %.3f\" % (cvs['test_r2'].std()))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The mean of the result is %.3f\" % (cvs['test_mae'].mean()))\nprint(\"The standard desviation error is %.3f\" % (cvs['test_mae'].std()))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = Test.drop(columns=['session_id','session_level'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = lr.predict(x)\noutput = pd.DataFrame({'Id': id,\n                      'session_level': preds.squeeze()})\n\noutput","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df = pd.read_csv('../input/predict-student-performance-from-game-play/sample_submission.csv')\nsample_submission_df['correct'] = lr.predict(x)\nsample_submission_df.to_csv('/kaggle/working/submission.csv', index=False)\nsample_submission_df.head()","metadata":{},"execution_count":null,"outputs":[]}]}