{"cells":[{"metadata":{},"cell_type":"markdown","source":"LIBARIES"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nimport gc\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.style as style\nstyle.use('fivethirtyeight')\n\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nfrom matplotlib.ticker import FuncFormatter\nimport plotly.express as px\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nimport riiideducation\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"LOAD DATA"},{"metadata":{},"cell_type":"raw","source":"TRAIN DATA"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"%%time\n\ndtypes = {\n    \"row_id\": \"int64\",\n    \"timestamp\": \"int64\",\n    \"user_id\": \"int32\",\n    \"content_id\": \"int16\",\n    \"content_type_id\": \"boolean\",\n    \"task_container_id\": \"int16\",\n    \"user_answer\": \"int8\",\n    \"answered_correctly\": \"int8\",\n    \"prior_question_elapsed_time\": \"float32\", \n    \"prior_question_had_explanation\": \"boolean\"\n}\n\ntrain = pd.read_csv(\"../input/riiid-test-answer-prediction/train.csv\", dtype=dtypes)\n\nprint(\"Train size:\", train.shape)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"VISULAZED FIRST 10 ENTRY OF DATA"},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head(10) #Visulazed first 10 rows of train dataset.\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Missing Values"},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Part of missing values for every column')\nprint(train.isnull().sum() / len(train))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Unique Users number"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'We have {train.user_id.nunique()} unique users in our train set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Question/Lecture Number"},{"metadata":{"trusted":true},"cell_type":"code","source":"#False=0 and means it is a question\n#True=1 and means it is a lecture\ntrain.content_type_id.value_counts() \n\"\"\"\ncontent_type_id denotes if the contents are questions or lectures.\nThe pie shows that 98.1% of the data in train.csv are questions (0), only 1.94% are lectures.\n\"\"\"\ndf = train['content_type_id'].value_counts().reset_index()\n\nfig = px.pie(df, values='content_type_id', names='index')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Unique content_id and how many of them is question"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'We have {train.content_id.nunique()} content ids in our train set, of which {train[train.content_type_id == False].content_id.nunique()} are batch of questions.')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Unique Task_container_id"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'We have {train.task_container_id.nunique()} unique Batches of questions or lectures.')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Answers Statistics"},{"metadata":{"trusted":true},"cell_type":"code","source":"train.user_answer.value_counts() #Count of answers\n\n\"\"\"\nuser_answer denotes if a user answered the question or not.\n0,1,2,3: I assume this means which option a user choose.\n-1: if content_type is lecture.\n\"\"\"\ndf = train['user_answer'].value_counts().reset_index()\n\nfig = px.pie(df, values='user_answer', names='index')\nfig.show()\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Most interactions are from users that were not active very long on the platform yet."},{"metadata":{"trusted":true},"cell_type":"code","source":"#1 year = 31536000000 ms\nts = train['timestamp']/(31536000000/12)\ntrain['timestamp']=ts.astype(np.int8)\nfig = plt.figure(figsize=(12,6))\nts.plot.hist(bins=100)\nplt.title(\"Histogram of timestamp\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Months between this user interaction and the first event completion from that user\")\nplt.show()\n\ndel ts\n\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Answered_correctly is our target, and we have to predict to probability for an answer to be correct. Without looking at the lecture interactions (-1), we see about 1/3 of the questions was answered incorrectly"},{"metadata":{"trusted":true},"cell_type":"code","source":"correct = train[train.answered_correctly != -1].answered_correctly.value_counts(ascending=True)\n\nfig = plt.figure(figsize=(12,4))\ncorrect.plot.barh()\nfor i, v in zip(correct.index, correct.values):\n    plt.text(v, i, '{:,}'.format(v), color='white', fontweight='bold', fontsize=14, ha='right', va='center')\nplt.title(\"Questions answered correctly\")\nplt.xticks(rotation=0)\nplt.show()\n\n\ndel correct\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Above I am plotting the number of answers per user_id against the percentage of questions answered correctly (sample of 200). As some users have answered huge amounts of questions, I have taken out the outliers (user_ids with 1000+ questions answered). As you can see, the trend is upward but there is also a lot of variation among users that have answered few questions."},{"metadata":{"trusted":true},"cell_type":"code","source":"user_percent = train[train.answered_correctly != -1].groupby('user_id')['answered_correctly'].agg(Mean='mean', Answers='count')\nprint(f'the highest number of questions answered by a user is {user_percent.Answers.max()}')\n\nuser_percent = user_percent.query('Answers <= 1000').sample(n=200, random_state=1)\n\nfig = plt.figure(figsize=(12,6))\nx = user_percent.Answers\ny = user_percent.Mean\nplt.scatter(x, y, marker='o')\nplt.title(\"Percent answered correctly versus number of questions answered\")\nplt.xticks(rotation=0)\nplt.xlabel(\"Number of questions answered\")\nplt.ylabel(\"Percent answered correctly\")\nz = np.polyfit(x, y, 1)\np = np.poly1d(z)\nplt.plot(x,p(x),\"r--\")\n\nplt.show()\n\n\ndel user_percent\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"To see that there is a relationship between prior_question_had_explanation and answered_correctly"},{"metadata":{"trusted":true},"cell_type":"code","source":"pq = train[train.answered_correctly != -1].groupby(['prior_question_had_explanation'], dropna=False).agg({'answered_correctly': ['mean', 'count']})\n#pq.index = pq.index.astype(str)\nprint(pq.iloc[:,1])\npq = pq.iloc[:,0]\n\nfig = plt.figure(figsize=(12,4))\npq.plot.barh()\n# for i, v in zip(pq.index, pq.values):\n#     plt.text(v, i, round(v,2), color='white', fontweight='bold', fontsize=14, ha='right', va='center')\nplt.title(\"Answered_correctly versus Prior Question had explanation\")\nplt.xlabel(\"Percent answered correctly\")\nplt.ylabel(\"Prior question had explanation\")\nplt.xticks(rotation=0)\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"number of questions acording to prior_question_had_explanation"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\nfreq = len(train)\n\ng = sns.countplot(train['prior_question_had_explanation'])\ng.set_title(\"Whether or not the user saw an explanation and the correct response (s) \\n after answering the previous question bundle\",\n            fontsize = 18)\ng.set_xlabel(\"prior_question_had_explanation\", fontsize = 15)\ng.set_ylabel(\"Count\", fontsize = 15)\n\nfor p in g.patches:\n    height = p.get_height()\n    g.text(p.get_x() + p.get_width() / 2., height + 3,\n          '{:1.2f}%'.format(height / freq * 100),\n          ha = \"center\", fontsize = 18)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Average Question alepsed time of true and false answers"},{"metadata":{"trusted":true},"cell_type":"code","source":"pq = train[train.answered_correctly != -1]\npq = pq[['prior_question_elapsed_time', 'answered_correctly']]\npq = pq.groupby(['answered_correctly']).agg({'answered_correctly': ['count'], 'prior_question_elapsed_time': ['mean']})\npq.head()\n\n\ndel pq\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Representing relation between total answered question and success"},{"metadata":{"trusted":true},"cell_type":"code","source":"\nusr_ans = train.groupby('user_id').agg({ 'answered_correctly': ['mean','sum', 'count']})\nusr_ans.columns = ['avg_correct_answer','num_of_correct', 'total_answers']\n\n# changing dtype for reducing memory (default = 64)\n\nusr_ans['num_of_correct'] = usr_ans['num_of_correct'].astype('int16')\nusr_ans['total_answers'] = usr_ans['total_answers'].astype('int16')\n\n\ntrain = pd.merge(train, usr_ans, how='left', on = 'user_id')\n\n# plotting total answers vs avg accuracy\n\nsns.regplot(data=usr_ans[usr_ans['total_answers']> 100], y='avg_correct_answer', x='total_answers', ci=False, scatter_kws={'alpha':0.5}, line_kws={\"color\": \"orange\"})\nplt.axhline(train.avg_correct_answer.mean(), color='k', linestyle='dashed', linewidth=3)\nplt.axvline(train.total_answers.mean(), color='k', linestyle='dashed', linewidth=3)\n\nmin_ylim, max_ylim = plt.ylim()\nplt.text(train.total_answers.mean()+25, max_ylim*0.20, 'Average Questions Solved {:.2f}'.format(train.total_answers.mean()))\nplt.text(train.total_answers.mean()+2400, max_ylim*0.65, 'Average Correct Answer: {:.2f}'.format(train.avg_correct_answer.mean()))\n\nplt.show()\n\ndel usr_ans\ngc.collect()\n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"QUESTION DATA"},{"metadata":{},"cell_type":"markdown","source":"General Visulasition"},{"metadata":{"trusted":true},"cell_type":"code","source":"questions.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"LECTURE DATA"},{"metadata":{},"cell_type":"markdown","source":"General Visulasition"},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%reset -f","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport matplotlib.style as style\nstyle.use('fivethirtyeight')\n\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nfrom matplotlib.ticker import FuncFormatter\nimport plotly.express as px\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nimport riiideducation\nimport optuna\nfrom optuna.samplers import TPESampler\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/riiid-test-answer-prediction'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\ndtypes = {\n    \"row_id\": \"int64\",\n    \"timestamp\": \"int64\",\n    \"user_id\": \"int32\",\n    \"content_id\": \"int16\",\n    \"content_type_id\": \"boolean\",\n    \"task_container_id\": \"int16\",\n    \"user_answer\": \"int8\",\n    \"answered_correctly\": \"int8\",\n    \"prior_question_elapsed_time\": \"float32\", \n    \"prior_question_had_explanation\": \"boolean\"\n}\n\ntrain = pd.read_csv(\"../input/riiid-test-answer-prediction/train.csv\", dtype=dtypes)\n\nprint(\"Train size:\", train.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"questions = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/questions.csv')\nlectures = pd.read_csv('/kaggle/input/riiid-test-answer-prediction/lectures.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"FEATURE ENGINEERING"},{"metadata":{},"cell_type":"markdown","source":" Split data into train data & feature engineering data (to use for past performance)\n Timestamp is in descending order - meaning that the last 10% observations have\n the biggest chance of having had some performance recorded before\n so looking at the performance in the past we'll try to predict the performance now"},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train['content_type_id'] != 1]\ntrain = train[train['answered_correctly'] != -1].reset_index(drop=True)\n\n\nfeatures_df = train.iloc[:int(9/10 * len(train))]\ntrain_df = train.iloc[int(9/10 * len(train)):]\n\n#del train\n#del questions\n#gc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Calculate mean, count, std with using groupby user_id/content_id"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n# --- STUDENT ANSWERS ---\n# Group by student\nuser_answers_df = features_df[features_df['answered_correctly']!=-1].\\\n                            groupby('user_id').\\\n                            agg({'answered_correctly': ['mean', 'count', 'std' ]}).\\\n                            reset_index()\n\nuser_answers_df.columns = ['user_id','mean_user_accuracy','questions_answered','std_user_accuracy']\n\n\n# --- CONTENT ID ANSWERS ---\n# Group by content\ncontent_answers_df = features_df[features_df['answered_correctly']!=-1].\\\n                            groupby('content_id').\\\n                            agg({'answered_correctly': ['mean', 'count', 'std' ]}).\\\n                            reset_index()\n\ncontent_answers_df.columns = ['content_id', 'mean_accuracy', 'question_asked', 'std_accuracy']\n\nuser_answers_df \ncontent_answers_df\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_answers_df ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"CREATE FEATURES AND TARGET SET"},{"metadata":{"trusted":true},"cell_type":"code","source":"features = [\n    'mean_user_accuracy', \n    'questions_answered',\n    'std_user_accuracy', \n\n    'mean_accuracy', \n    'question_asked',\n    'std_accuracy', \n    \n    'prior_question_elapsed_time', \n    'prior_question_had_explanation',\n]\n\ntarget = 'answered_correctly'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"MERGE NEW FEATURES     acıkla ve mıssıng value tablosunu basa ekle kesın"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\ntrain_df = train_df.merge(user_answers_df, how='left', on='user_id')\ntrain_df = train_df.merge(content_answers_df, how='left', on='content_id')\n\ntrain_df = train_df[features + [target]]\n\nlabel_enc = LabelEncoder()\n\ntrain_df['prior_question_had_explanation'].fillna(False, inplace = True)\ntrain_df['prior_question_had_explanation'] = label_enc.fit_transform(train_df['prior_question_had_explanation'])\n\nmean_prior = train_df.prior_question_elapsed_time.astype(\"float64\").mean()\n\ntrain_df['prior_question_elapsed_time'].fillna(mean_prior, inplace = True)\n\n\ntrain_df = train_df.replace([np.inf, -np.inf], np.nan)\ntrain_df = train_df.fillna(0.5)\n\ntrain_df\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"TRAIN TEST SPLIT"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, test_df = train_test_split(train_df, random_state = 123, test_size = 0.2)\n\nx_train = train_df[features]\ny_train = train_df[target]\nx_test = test_df[features]\ny_test = test_df[target]\n\ndel train_df\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"LOGISTIC REGRESSION"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nmodel_LR = LogisticRegression()\n\nmodel_LR.fit(x_train, y_train)\n\nns_probs = [0 for _ in range(len(y_test))]\n\npredictions_LR =  model_LR.predict_proba(x_test)\n\n# Keep only positive outcomes\nLR_probs = predictions_LR[:, 1]\n\nns_auc = roc_auc_score(y_test, ns_probs)\nLR_auc = roc_auc_score(y_test, predictions_LR[:,1])\n\nprint('Logistic: ROC AUC = %.3f' % (LR_auc))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from matplotlib import pyplot\n# calculate roc curves\nns_fpr, ns_tpr, _ = roc_curve(y_test, ns_probs)\nLR_fpr, LR_tpr, _ = roc_curve(y_test, LR_probs)\n\n# figure size\nplt.rcParams[\"figure.figsize\"] = (9, 5)\n\n# plot the roc curve for the model\npyplot.plot(ns_fpr, ns_tpr, linestyle = '--', label = 'No Skill')\npyplot.plot(LR_fpr, LR_tpr, linestyle = '-', label = 'Logistic')\n\n# axis labels\npyplot.xlabel('False Positive Rate', fontsize = 15)\npyplot.ylabel('True Positive Rate', fontsize = 15)\n\n# show the legend\npyplot.legend(fontsize = 15)\n\n# show the plot\npyplot.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"XGBM"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nmodel_XGB = XGBClassifier()\n\nmodel_XGB.fit(x_train, y_train)\n\nprediction_XGB = model_XGB.predict_proba(x_test)\n\n# keep probabilities for the positive outcome only\nXGB_probs = prediction_XGB[:, 1]\n\nns_auc = roc_auc_score(y_test, ns_probs)\nXGB_auc = roc_auc_score(y_test, prediction_XGB[:,1])\n\nprint('XGBoost: ROC AUC = %.3f' % (XGB_auc))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# calculate roc curves\nns_fpr, ns_tpr, _ = roc_curve(y_test, ns_probs)\nXGB_fpr, XGB_tpr, _ = roc_curve(y_test, XGB_probs)\n\n# figure size\nplt.rcParams[\"figure.figsize\"] = (9, 5)\n\n# plot the roc curve for the model\npyplot.plot(ns_fpr, ns_tpr, linestyle = '--', label = 'No Skill')\npyplot.plot(XGB_fpr, XGB_tpr, linestyle = '-', label = 'XGBoost', color = \"red\")\n\n# axis labels\npyplot.xlabel('False Positive Rate', fontsize = 15)\npyplot.ylabel('True Positive Rate', fontsize = 15)\n\n# show the legend\npyplot.legend(fontsize = 15)\n\n# show the plot\npyplot.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"LGBM"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nmodel_LGBM = LGBMClassifier()\n\nmodel_LGBM.fit(x_train, y_train)\n\nprediction_LGBM = model_LGBM.predict_proba(x_test)\nns_probs = [0 for _ in range(len(y_test))]\n\n# keep probabilities for the positive outcome only\nLGBM_probs = prediction_LGBM[:, 1]\n\nns_auc = roc_auc_score(y_test, ns_probs)\nLGBM_auc = roc_auc_score(y_test, prediction_LGBM[:,1])\n\nprint('LGBM: ROC AUC = %.3f' % (LGBM_auc))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from matplotlib import pyplot\n# calculate roc curves\nns_fpr, ns_tpr, _ = roc_curve(y_test, ns_probs)\nLGBM_fpr, LGBM_tpr, _ = roc_curve(y_test, LGBM_probs)\n\n# figure size\nplt.rcParams[\"figure.figsize\"] = (9, 5)\n\n# plot the roc curve for the model\npyplot.plot(ns_fpr, ns_tpr, linestyle = '--', label = 'No Skill')\npyplot.plot(LGBM_fpr, LGBM_tpr, linestyle = '-', label = 'LGBM', color = \"green\")\n\n# axis labels\npyplot.xlabel('False Positive Rate', fontsize = 15)\npyplot.ylabel('True Positive Rate', fontsize = 15)\n\n# show the legend\npyplot.legend(fontsize = 12)\n\n# show the plot\npyplot.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"OPTUNA"},{"metadata":{"trusted":true},"cell_type":"code","source":"import optuna\nfrom optuna.samplers import TPESampler\n\nsampler = TPESampler(\n    seed=666\n)\n\n\n\n\ndef create_model(trial):\n    num_leaves = trial.suggest_int(\"num_leaves\", 2, 31)\n    n_estimators = trial.suggest_int(\"n_estimators\", 50, 300)\n    max_depth = trial.suggest_int('max_depth', 3, 8)\n    min_child_samples = trial.suggest_int('min_child_samples', 100, 1200)\n    learning_rate = trial.suggest_uniform('learning_rate', 0.0001, 0.99)\n    min_data_in_leaf = trial.suggest_int('min_data_in_leaf', 5, 90)\n    bagging_fraction = trial.suggest_uniform('bagging_fraction', 0.0001, 1.0)\n    feature_fraction = trial.suggest_uniform('feature_fraction', 0.0001, 1.0)\n    \n    model = LGBMClassifier(\n        num_leaves=num_leaves,\n        n_estimators=n_estimators, \n        max_depth=max_depth, \n        min_child_samples=min_child_samples, \n        min_data_in_leaf=min_data_in_leaf,\n        learning_rate=learning_rate,\n        feature_fraction=feature_fraction,\n        random_state=666\n    )\n    return model\n\n\ndef objective(trial):\n    model = create_model(trial)\n    model.fit(x_train, y_train)\n    score = roc_auc_score(\n        y_test.values, \n        model.predict_proba(x_test)[:,1]\n    )\n    return score\n\nstudy = optuna.create_study(direction=\"maximize\", sampler=sampler)\nstudy.optimize(objective, n_trials=10)\nparams = study.best_params\nparams['random_state'] = 666\n    \nmodel = LGBMClassifier(\n    **params\n)\n\nmodel.fit(x_train, y_train)\nprint('LGB score: ', roc_auc_score(y_test, model.predict_proba(x_test)[:,1]))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"LGBM WITH PARAMETERS ('objective': 'binary', 'metric': 'auc' were set by us)"},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {'num_leaves': 21,\n          'n_estimators': 110,\n          'max_depth': 7,\n          'min_child_samples': 882,\n          'learning_rate': 0.4582669848936411,\n          'min_data_in_leaf': 21,\n          'bagging_fraction': 0.200327514580544,\n          'feature_fraction': 0.7441797533972156,\n          'objective': 'binary',\n          'metric': 'auc'}\nlastmodel = LGBMClassifier(**params)\nlastmodel.fit(x_train, y_train)\nprint('LGB score: ', roc_auc_score(y_test, lastmodel.predict_proba(x_test)[:,1]))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"FEATURE IMPORTANCE"},{"metadata":{"trusted":true},"cell_type":"code","source":"feat_importance = pd.DataFrame()\nfeat_importance[\"feature\"] = x_train.columns\nfeat_importance[\"value\"] = lastmodel.feature_importances_\nfeat_importance.sort_values(by='value', ascending=False, inplace=True)\n\nplt.figure(figsize=(8,10))\nax = sns.barplot(y=\"feature\", x=\"value\", data=feat_importance)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"env = riiideducation.make_env()\niter_test = env.iter_test()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for (test_df, sample_prediction_df) in iter_test:\n    \n    test_df = pd.merge(test_df, questions, left_on='content_id', right_on='question_id', how='left')\n    test_df.drop(columns=['question_id'], inplace=True)\n    test_df['prior_question_had_explanation'].fillna(bool(True), inplace=True)\n    test_df = test_df.replace([-np.inf, np.inf], np.nan)\n    test_df = test_df.fillna(test_df.mean())\n    \n    test_df = test_df[test_df['content_type_id'] != 1]\n    \n    test_df = test_df.merge(user_answers_df, how='left', on='user_id')\n    test_df = test_df.merge(content_answers_df, how='left', on='content_id')\n    test_df['prior_question_had_explanation'] = test_df['prior_question_had_explanation'].astype(bool)\n    test_df = test_df.replace([-np.inf, np.inf], np.nan)\n    test_df = test_df.fillna(test_df.mean())\n\n    \n    test_df['answered_correctly'] = lastmodel.predict_proba(test_df[features])[:,1]\n    env.predict(test_df.loc[test_df['content_type_id'] == 0, ['row_id', 'answered_correctly']])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}