{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"\n#important links\n## https://github.com/riiid/ednet","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import optuna\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom catboost import CatBoostClassifier\nfrom  sklearn.tree import DecisionTreeClassifier\nfrom  sklearn.model_selection import train_test_split\nimport operator\nimport random\n\n# visualize\nimport matplotlib.pyplot as plt\nimport matplotlib.style as style\nimport seaborn as sns\nfrom matplotlib import pyplot\nfrom matplotlib.ticker import ScalarFormatter\nsns.set_context(\"talk\")\nstyle.use('fivethirtyeight')\n\nimport riiideducation\nimport dask.dataframe as dd\nimport  pandas as pd\nimport numpy as np\nfrom sklearn.metrics import roc_auc_score\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train_df= pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv',\n#                 usecols=[1, 2, 3,4,7,8,9],nrows=10**6, dtype={'timestamp': 'int64', 'user_id': 'int32' ,\n#                                                   'content_id': 'int16','content_type_id': 'int8',\n#                                                   'answered_correctly':'int8',\n#                                                   'prior_question_elapsed_time': 'float32',\n#                                                   'prior_question_had_explanation': 'boolean'}\n#               )\ntrain_df= pd.read_csv('/kaggle/input/riiid-test-answer-prediction/train.csv',\n                nrows=10**6, dtype={'timestamp': 'int64', 'user_id': 'int32' ,\n                                                  'content_id': 'int16','content_type_id': 'int8',\n                                    'task_container_id':'int16','user_answer':'int8',\n                                                  'answered_correctly':'int8',\n                                                  'prior_question_elapsed_time': 'float32',\n                                                  'prior_question_had_explanation': 'boolean'}\n              )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp = data[data.user_id == np.random.choice(data.user_id.unique())].sort_values(\"timestamp\")\nprint (temp.shape)\ntemp.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# # EDA"},{"metadata":{},"cell_type":"markdown","source":"1. Time stamp - basically how much time passed from the first question complete and next question started.\n         how much time he took, how many question he attempt in a day , like that"},{"metadata":{"trusted":true},"cell_type":"code","source":"data['timestamp'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# checking null\ndata['timestamp'].isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f = plt.figure(figsize=(16, 8))\ngs = f.add_gridspec(1, 2)\n\nwith sns.axes_style(\"whitegrid\"):\n    ax = f.add_subplot(gs[0, 0])\n    data['timestamp'].hist(bins = 50,color='orange')\n    plt.title(\"Timestamp Distribution\")\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"a lot of user a taking a lot of time or gap between, let order by timestamp and replace it by subtracting from previous timestamp "},{"metadata":{"trusted":true},"cell_type":"code","source":"data = data.sort_values(['user_id','timestamp'])\ndata.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = data.sort_values(['user_id','task_container_id'])\ndata.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def countplot(column):\n    plt.figure(dpi=100)\n    sns.countplot(train_data[column])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# sns.distplot(data['timestamp'],color='yellow')\n# plt.show()\ndata['timestamp'].hist()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# task contaioner id not in increasing order \ndata['task_container_id'] = (\n    data\n    .groupby('user_id')['task_container_id']\n    .transform(lambda x: pd.factorize(x)[0])\n    .astype('int16')\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['time_elapsed'] = data.groupby('user_id')['timestamp'].apply(lambda x: x- x.shift(1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['time_elapsed'] = data['time_elapsed'].fillna(data.groupby('user_id')['time_elapsed'].transform('mean'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"time_correct = data[data['answered_correctly']==1]['task_container_id']\ntime_wrong = data[data['answered_correctly']!=1]['task_container_id']\n# https://glowingpython.blogspot.com/2012/09/boxplot-with-matplotlib.html\nplt.boxplot([time_correct, time_wrong])\nplt.xticks([1,2],('Approved Projects','Rejected Projects'))\nplt.ylabel('Words in project title')\nplt.grid()\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,3))\nsns.kdeplot(time_correct ,label=\"correct ans\", bw=0.6)\nsns.kdeplot(time_wrong,label=\"wrong ans\", bw=0.6)\nplt.title('time stamp')\nplt.xlabel('')\nplt.legend()\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"time_correct = data[data['answered_correctly']==1]['time_elapsed']\ntime_wrong = data[data['answered_correctly']!=1]['time_elapsed']\nplt.figure(figsize=(10,3))\nsns.kdeplot(time_correct ,label=\"correct ans\", bw=0.6)\nsns.kdeplot(time_wrong,label=\"wrong ans\", bw=0.6)\nplt.title('time stamp')\nplt.xlabel('')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_by_dv(data,col):\n    time_correct = data[data['answered_correctly']==1][col]\n    time_wrong = data[data['answered_correctly']!=1][col]\n    plt.figure(figsize=(10,3))\n    sns.kdeplot(time_correct ,label=\"correct ans\", bw=0.6)\n    sns.kdeplot(time_wrong,label=\"wrong ans\", bw=0.6)\n    plt.title('time stamp')\n    plt.xlabel('')\n    plt.legend()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# User ID"},{"metadata":{"trusted":true},"cell_type":"code","source":"# unique user id \nprint(\"total data\",len(data),len(data['user_id'].unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['total_question_attemp'] = data.groupby('user_id')['user_id'].transform('count')\ndata['total_question_attemp_correct'] = data.groupby('user_id')['answered_correctly'].transform('sum')\ndata['total_question_attemp_correct'] = data['total_question_attemp_correct']/data['total_question_attemp']\ndata.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_by_dv(data,'total_question_attemp')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_by_dv(data,'total_question_attemp_correct')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# content_id"},{"metadata":{"trusted":true},"cell_type":"code","source":"# unique question  \nprint(\"total data\",len(data),len(data['content_id'].unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['question_attemp'] = data.groupby('content_id')['content_id'].transform('count')\ndata['question_ans_correct'] = data.groupby('content_id')['answered_correctly'].transform('sum')\ndata['question_ans_correct'] = data['question_ans_correct']/data['question_attemp']\ndata.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_by_dv(data,'question_attemp')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_by_dv(data,'question_ans_correct')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# content_type_id"},{"metadata":{"trusted":true},"cell_type":"code","source":"def stack_plot(data, xtick, col2, col3='total'):\n     ind = np.arange(data.shape[0])\n\n     plt.figure(figsize=(20,5))\n     p1 = plt.bar(ind, data[col3].values)\n     p2 = plt.bar(ind, data[col2].values)\n     plt.ylabel('Projects')\n     plt.title('Number of projects aproved vs rejected')\n     plt.xticks(ind, list(data[xtick].values))\n     plt.legend((p1[0], p2[0]), ('total', 'accepted'))\n     plt.show()\n\n\ndef univariate_barplots(data, col1, col2, top=False):\n\ttemp = pd.DataFrame(data.groupby(col1)[col2].agg(lambda x: x.eq(1).sum())).reset_index()\n\t# Pandas dataframe grouby count: https://stackoverflow.com/a/19385591/4084039\n\ttemp['total'] = pd.DataFrame(data.groupby(col1)[col2].agg(total='count')).reset_index()['total']\n\ttemp['Avg'] = pd.DataFrame(data.groupby(col1)[col2].agg(Avg='mean')).reset_index()['Avg']\n\n\ttemp.sort_values(by=['total'],inplace=True, ascending=False)\n\n\tif top:\n\t\ttemp = temp[0:top]\n\n\tstack_plot(temp, xtick=col1, col2=col2, col3='total')\n\tprint(temp.head(5))\n\tprint(\"=\"*50)\n\t#print(temp.tail(5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.groupby(['content_type_id','answered_correctly']).agg('count').iloc[:,:1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data[data['content_type_id']==1].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## removing latures data\ndata = data[data['content_type_id']!=1]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# task_container_id"},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['task_container_id'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# user_answer"},{"metadata":{"trusted":true},"cell_type":"code","source":"data['user_answer'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# answered_correctly"},{"metadata":{"trusted":true},"cell_type":"code","source":"data['answered_correctly'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# prior_question_elapsed_time"},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_elapsed_time'].isna().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_elapsed_time'] = data['prior_question_elapsed_time'].fillna(data.groupby('user_id')['prior_question_elapsed_time'].transform('mean'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_by_dv(data,'prior_question_elapsed_time')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# prior_question_had_explanation"},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_had_explanation'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_had_explanation'].isnaprior_question_had_explanation().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['prior_question_had_explanation'] = \\\ndata['prior_question_had_explanation'].fillna(data['prior_question_had_explanation'].mode()[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Feature engineering"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train.content_type_id == False]\n#arrange by timestamp\ntrain = train.sort_values(['timestamp'], ascending=True)\n\ntrain.drop(['timestamp','content_type_id'], axis=1,   inplace=True)\n\nresults_c = train[['content_id','answered_correctly']].groupby(['content_id']).agg(['mean'])\nresults_c.columns = [\"answered_correctly_content\"]\n\nresults_u = train[['user_id','answered_correctly']].groupby(['user_id']).agg(['mean', 'sum'])\nresults_u.columns = [\"answered_correctly_user\", 'sum']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train.iloc[:,:]\nX = pd.merge(X, results_u, on=['user_id'], how=\"left\")\nX = pd.merge(X, results_c, on=['content_id'], how=\"left\")\nX=X[X.answered_correctly!= -1 ]\nX=X.sort_values(['user_id'])\nY = X[[\"answered_correctly\"]]\nX = X.drop([\"answered_correctly\"], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# fill_mode = lambda col: col.fillna(col.mode())\n# X = X.apply(fill_mode, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X[\"prior_question_had_explanation\"].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlb_make = LabelEncoder()\nX['prior_question_had_explanation'].fillna(X['prior_question_had_explanation'].mode()[0], inplace=True)\nX['prior_question_elapsed_time'].fillna(X['prior_question_elapsed_time'].mean(), inplace=True)\n\nX[\"prior_question_had_explanation_enc\"] = lb_make.fit_transform(X[\"prior_question_had_explanation\"])\nX.head()\n\nX = X[['answered_correctly_user', 'answered_correctly_content', 'sum','prior_question_elapsed_time','prior_question_had_explanation_enc']] \n#X.fillna(0.5,  inplace=True)\n\nXt, Xv, Yt, Yv = train_test_split(X, Y, test_size = 0.2, shuffle=False, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"XGBoost version:\", xgb.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import RandomizedSearchCV\n\nx_cfl=XGBClassifier(objective='binary:logistic',eval_metric= 'auc',tree_method = 'gpu_hist',\n                    n_jobs=-1)\n\nprams={\n    'learning_rate':[0.01,0.03,0.05,0.1,0.15,0.2],\n     'n_estimators':[100,200,500,1000,2000,3000,4000],\n     'max_depth':[3,5,10],\n    'colsample_bytree':[0.1,0.3,0.5,1],\n    'subsample':[0.1,0.3,0.5,1]\n}\n\nrandom_cfl=RandomizedSearchCV(x_cfl,param_distributions=prams,verbose=10,n_jobs=-1,cv=3)\nrandom_cfl.fit(Xt, Yt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_pred = x_cfl.predict(Xv)\n    \n# CV score\nscore = roc_auc_score(Yv, val_pred)\nprint(f\"AUC = {score}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}