{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"from matplotlib import pyplot as plt \n%matplotlib inline\nimport seaborn as sns\n#\nfrom sklearn.metrics import accuracy_score\n#","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"big_data_train = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv', nrows=100000)\nquestions_df = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures_df = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\ntest_df = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = big_data_train[['user_id', 'content_id', 'content_type_id', 'task_container_id', 'prior_question_elapsed_time', 'prior_question_had_explanation']]\ny = big_data_train[['answered_correctly']]\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n#\nX_train,X_test,y_train,y_test = train_test_split(X,y,test_size=0.33)\nX_train.shape, X_test.shape, y_train.shape, y_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.experimental import (enable_iterative_imputer,)\nfrom sklearn import impute\n#\nnum_cols = [\"user_id\", \"content_id\", \"content_type_id\", \"task_container_id\", \"prior_question_elapsed_time\", \"prior_question_had_explanation\"]\nimputer = impute.IterativeImputer()\nimputed = imputer.fit_transform(X_train[num_cols])\nX_train.loc[:, num_cols] = imputed\n#X_train\nimputed = imputer.fit_transform(X_test[num_cols])\nX_test.loc[:, num_cols] = imputed\nX_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\nfrom sklearn.ensemble import RandomForestClassifier\n#\nrfclf = RandomForestClassifier()\nrfclf.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred=rfclf.predict(X_train)\nt_pred=rfclf.predict(X_test)\nprint(accuracy_score(y_pred, y_train))\nprint(accuracy_score(t_pred, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#params\nfrom sklearn.model_selection import cross_validate, GridSearchCV\n#\nparams = {\n    'max_features':[0.4, \"auto\"],\n    'n_estimators':[15, 200],\n    'min_samples_leaf':[1, 0.1],\n    'random_state':[10, 42, 100]\n}\n#\ncv = GridSearchCV(\n    rfclf,\n    params,\n    n_jobs = -1\n).fit(X_train,y_train)\ncv.best_params_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rfc_m = RandomForestClassifier(\n    **{'max_features': 0.4,\n       'min_samples_leaf': 0.1,\n       'n_estimators': 15,\n       'random_state': 10,\n      }\n    )\nrfc_m.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred=rfc_m.predict(X_train)\nt_pred=rfc_m.predict(X_test)\nprint(accuracy_score(y_pred, y_train))\nprint(accuracy_score(t_pred, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import xgboost as xgb\n#\nxgb_m = xgb.XGBClassifier()\n#xgbclf = xgb.XGBClassifier(random_state=42)\n#xgbclf.fit(X_train,y_train, early_stopping_rounds=10, eval_set=[(X_test, y_test)],)\nxgb_m.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred=xgb_m.predict(X_train)\nt_pred=xgb_m.predict(X_test)\nprint(accuracy_score(y_pred, y_train))\nprint(accuracy_score(t_pred, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\nfor col, val in sorted(\n    zip(\n        X.columns,\n        xgb_m.feature_importances_,\n    ),\n    key=lambda x: x[1],\n    reverse=True,\n)[:5]:\n    print(f\"{col:10}{val:10.3f}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\nfig, ax = plt.subplots(figsize=(6, 4))\nxgb.plot_importance(xgb_m, ax=ax)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\nbooster = xgb_m.get_booster()\nprint(booster.get_dump()[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.neural_network import MLPClassifier\n#\nmlp_m = MLPClassifier()\nmlp_m.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred=mlp_m.predict(X_train)\nt_pred=mlp_m.predict(X_test)\nprint(accuracy_score(y_pred, y_train))\nprint(accuracy_score(t_pred, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#lgbm\nimport lightgbm as lgb","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\nlgb_m = lgb.LGBMClassifier(random_state=42)\n#lgb_m = lgb.LGBMClassifier()\nlgb_m.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred=lgb_m.predict(X_train)\nt_pred=lgb_m.predict(X_test)\nprint(accuracy_score(y_pred, y_train))\nprint(accuracy_score(t_pred, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#\nfor col, val in sorted(\n    zip(\n        X.columns,\n        lgb_m.feature_importances_,\n    ),\n    key=lambda x: x[1],\n    reverse=True,\n)[:5]:\n    print(f\"{col:10}{val:10.3f}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(6, 4))\nlgb.plot_importance(lgb_m, ax=ax)\nfig.tight_layout()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}