{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\n    '/kaggle/input/riiid-test-answer-prediction/train.csv',\n    low_memory=False,\n    nrows=10**6,\n    usecols=[  \n        'user_id',\n        'content_id',\n        'content_type_id',\n        'answered_correctly',\n        'prior_question_elapsed_time',\n        'prior_question_had_explanation'\n    ],\n       dtype={\n         \n                'user_id': 'int32',\n                'content_id': 'int16',\n                'content_type_id': 'int8',\n                'answered_correctly': 'int8',\n                'prior_question_elapsed_time': 'float32',\n                'prior_question_had_explanation': 'boolean'\n       }\n    \n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"para=['user_id',\n        'content_id',\n        'content_type_id',\n        'prior_question_elapsed_time',\n        'prior_question_had_explanation']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=train_df[train_df['answered_correctly']!=-1]\ntrain_df = train_df[train_df['content_type_id'] == 0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"group_user_id=train_df.groupby('user_id')\ngroup_content_id=train_df.groupby('content_id')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"user_answers_df = group_user_id.agg({'answered_correctly': ['mean', 'count', 'std', 'median', 'skew','var']}).copy()\nuser_answers_df.columns = [\n    'mean_user_accuracy', \n    'questions_answered', \n    'std_user_accuracy', \n    'median_user_accuracy', \n    'skew_user_accuracy',\n    'var_accuracy_user'\n]\nuser_answers_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cont_answers_df = group_content_id.agg({'answered_correctly': ['mean', 'count', 'std', 'median', 'skew','var']}).copy()\ncont_answers_df.columns = [\n    'mean_cont_accuracy', \n    'cont_questions_answered', \n    'std_cont_accuracy', \n    'median_cont_accuracy', \n    'skew_cont_accuracy',\n    'var_accuracy_content'\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=train_df.merge(user_answers_df, how='left', on='user_id')\ntrain_df=train_df.merge(cont_answers_df, how='left', on='content_id')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\n\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df.replace([np.inf, -np.inf], np.nan)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df=train_df.fillna(method='bfill')\ntrain_df=train_df.fillna(method='ffill')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['prior_question_had_explanation']=train_df['prior_question_had_explanation'].astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from pandas import DataFrame\nfrom pandas import concat\n\n\ndef series_to_supervised(data, n_in=1, n_out=1, dropnan=True):\n\n    n_vars = 1 if type(data) is list else data.shape[1]\n    df = DataFrame(data)\n    cols, names = list(), list()\n    # input sequence (t-n, ... t-1)\n    for i in range(n_in, 0, -1):\n        cols.append(df.shift(i))\n        names += [('var%d(t-%d)' % (j+1, i)) for j in range(n_vars)]\n    # forecast sequence (t, t+1, ... t+n)\n    for i in range(0, n_out):\n        cols.append(df.shift(-i))\n        if i == 0:\n            names += [('var%d(t)' % (j+1)) for j in range(n_vars)]\n        else:\n            names += [('var%d(t+%d)' % (j+1, i)) for j in range(n_vars)]\n    # put it all together\n    agg = concat(cols, axis=1)\n    agg.columns = names\n    # drop rows with NaN values\n    #if fillna:\n        #agg.fillna(0)\n    return agg\n\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"items=['prior_question_elapsed_time', 'prior_question_had_explanation',\n       'mean_user_accuracy', 'questions_answered', 'std_user_accuracy',\n       'median_user_accuracy', 'skew_user_accuracy', 'var_accuracy_user',\n       'mean_cont_accuracy', 'cont_questions_answered', 'std_cont_accuracy',\n       'median_cont_accuracy', 'skew_cont_accuracy', 'var_accuracy_content'\n    ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nlabels = train_df['answered_correctly']\nfeatures = train_df[items]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"features['prior_question_elapsed_time']=features['prior_question_elapsed_time']/features['prior_question_elapsed_time'].max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndata= series_to_supervised(features,2,2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data= data.replace([np.inf, -np.inf], np.nan)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data=data.fillna(method='bfill')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data=data.fillna(method='ffill')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(data, labels, \n                                                                            test_size = 0.2, random_state = 42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_features.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print ( data.shape)\nprint(labels.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from lightgbm import LGBMClassifier\nlgbm=LGBMClassifier(boosting_type='gbdt', objective='binary', num_leaves=50,\n                                learning_rate=0.1, n_estimators=400, max_depth=-1,\n                                bagging_fraction=0.9, feature_fraction=0.9, reg_lambda=0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lgbm.fit(train_features,  train_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_out=lgbm.predict_proba(test_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#y_score=y_out[:,1]\ny_score1=y_out[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import roc_curve\n#fpr_clf, tpr_clf, threshold_clf=roc_curve(labels, y_score)\nfpr_clf1, tpr_clf1, threshold_clf1=roc_curve(test_labels, y_score1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_roc_curve(fpr, tpr, label=None):\n    plt.plot(fpr, tpr, linewidth=2, label=label)\n    plt.plot([0, 1], [0, 1], 'k--')\n    plt.axis([0, 1, 0, 1])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n#plot_roc_curve(fpr_clf, tpr_clf,'Random Forest')\nplot_roc_curve(fpr_clf1, tpr_clf1,'Random_Forest')\nplt.legend(loc='bottom right')\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\nCval=roc_auc_score(test_labels, y_score1)\nCval","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dtest=pd.read_csv('/kaggle/input/riiid-test-answer-prediction/example_test.csv')\ndtest.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n    dimap=[]\n    dimav=[]\n    dimav.append(dtest.shape)\n    dtest = dtest[dtest['content_type_id'] == 0]\n    dtest=dtest.merge(user_answers_df, how='left', on='user_id')\n    dtest=dtest.merge(cont_answers_df, how='left', on='content_id') \n    dtest= dtest.replace([np.inf, -np.inf], np.nan)\n    dtest=dtest.fillna(method='bfill')\n    dtest['prior_question_had_explanation']=dtest['prior_question_had_explanation'].astype(int)\n    test= dtest[items]\n    test= series_to_supervised(test,2,2)\n    test=test.fillna(0)\n    y_preds = lgbm.predict_proba(test)\n    dtest['answered_correctly'] = y_preds[:,1]\n    dimap.append(dtest.shape)\n    response=dtest.loc[dtest['content_type_id'] == 0, ['row_id', 'answered_correctly']]\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import riiideducation \nenv = riiideducation.make_env()\niter_test = env.iter_test()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x=[]\nyav=[]\nyap=[]\ndimap=[]\ndimav=[]\nymed=[]\nfor (test_df, sample_prediction_d) in iter_test:\n    test_df = test_df[test_df['content_type_id'] == 0]\n    dimav.append(test_df.shape)\n    test_d=test_df[para]\n    test_d=test_d.merge(user_answers_df, how='left', on='user_id')\n    test_d=test_d.merge(cont_answers_df, how='left', on='content_id') \n    ymed.append(test_d)\n    test_d = test_d.replace([np.inf, -np.inf], np.nan)\n    test_d=test_d.fillna(method='bfill')\n    test_d=test_d.fillna(method='ffill')\n    test_d['prior_question_had_explanation']=test_d['prior_question_had_explanation'].astype(int)\n    test= test_d[items]\n    test= series_to_supervised(test,2,2)\n    \n    test=test.fillna(0)\n    ymed.append(test)\n    y_preds = lgbm.predict_proba(test)\n    x.append(y_preds[:,1])\n    test_df['answered_correctly'] = y_preds[:,1]\n    dimap.append(test_df.shape)\n    yap.append(test_df)\n    env.predict(test_df.loc[:,['row_id', 'answered_correctly']])\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dimap","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"yap[2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}