{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# installing datatable\n!pip install datatable","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Feature Engineering"},{"metadata":{"trusted":true},"cell_type":"code","source":"questions = pd.read_csv('../input/riiid-test-answer-prediction/questions.csv',\n                       dtype = {\n                           'question_id':'int64',\n                           'bundle_id':'int64',\n                           'correct_answer':'object',\n                           'part':'int64'})\nquestions.fillna('-1', inplace = True)\n\nimport nltk, re\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom sklearn.cluster import KMeans\n\ndef tokenize_and_stem(text):\n    tokens = [word for sent in nltk.sent_tokenize(text) for word in nltk.word_tokenize(sent)]\n    filtered_tokens = []\n    for token in tokens:\n        if re.search('[0-9]', token):\n            filtered_tokens.append(token)\n    return filtered_tokens\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\n#define vectorizer parameters\ntfidf_vectorizer = TfidfVectorizer(max_df=1.0, max_features=200000,\n                                 min_df=0.0,use_idf=True, tokenizer=tokenize_and_stem, ngram_range=(1,3))\n%time tfidf_matrix = tfidf_vectorizer.fit_transform(questions.tags) #fit the vectorizer to synopses\nprint(tfidf_matrix.shape)\ntags = tfidf_vectorizer.get_feature_names()\ndist = 1 - cosine_similarity(tfidf_matrix)\n\nNUM_CLUSTERS = 7\n\n\nkm = KMeans(n_clusters=NUM_CLUSTERS)\n\n%time km.fit(tfidf_matrix)\n\nclusters = km.labels_.tolist()\n\nn_words = 5\n\norder_centroids = km.cluster_centers_.argsort()[:, ::-1] \n\nfor i in range(NUM_CLUSTERS):\n    print(\"Cluster %d words:\" % i, end='')\n    \n    for ind in order_centroids[i, :n_words]: \n        print(' %s' % tags[ind], end=' ')\n    print() #add whitespace\n    \nquestions['kmean_cluster'] = clusters\n\nquestions.to_csv('./questions_fe.csv', index = False) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lectures = pd.read_csv('../input/riiid-test-answer-prediction/lectures.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import datatable as dt\ndf = dt.fread(\"../input/riiid-test-answer-prediction/train.csv\")\n#df = pd.read_csv('../input/riiid-test-answer-prediction/train.csv',nrows = 1000000)\n#df = df[df['content_type_id']!=1].merge(questions[['question_id', 'bundle_id','part','kmean_cluster']], left_on='content_id', right_on = 'question_id')\n#df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.names = {\"content_id\": \"question_id\"}\nquestions  = dt.Frame(questions[['question_id', 'bundle_id','part','kmean_cluster']])\nquestions.key = \"question_id\"\ndf = df[dt.f.content_type_id == 0, :]\ndf = df[: , : , dt.join(questions)]\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from datatable import (dt, f, by, ifelse, update, sort,\n                      count, min, max, mean, sum, rowsum,sd, unique)\nnum_users = unique(df[:,dt.f.user_id]).shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bundle_stats = df[:,{'mean': mean(dt.f.answered_correctly), 'std': sd(dt.f.answered_correctly),'nunique':count(dt.f.user_id)/num_users, 'count':count(dt.f.row_id)}, by('bundle_id')]\nbundle_stats = bundle_stats.to_pandas()\nbundle_stats.columns = ['bundle_id','bundle_mean_answered_correctly','bundle_std_answered_correctly','bundle_perc_students','bundle_times']\nbundle_stats.to_csv('./bundle_stats.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import gc \ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"part_stats = df[:,{\n    'mean': mean(dt.f.answered_correctly), \n    'std': sd(dt.f.answered_correctly),\n    'nunique':count(dt.f.user_id)/num_users, \n    'count':count(dt.f.row_id)}, \n                by('part')]\npart_stats = part_stats.to_pandas()\npart_stats.columns = ['part','part_mean_answered_correctly','part_std_answered_correctly','part_perc_students','part_times']\npart_stats.to_csv('./part_stats.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cluster_stats = df[:,{\n    'mean': mean(dt.f.answered_correctly), \n    'std': sd(dt.f.answered_correctly),\n    'nunique':count(dt.f.user_id)/num_users, \n    'count':count(dt.f.row_id)}, \n                by('kmean_cluster')]\ncluster_stats = cluster_stats.to_pandas()\ncluster_stats.columns = ['kmean_cluster','cluster_mean_answered_correctly','cluster_std_answered_correctly','cluster_perc_students','cluster_times']\ncluster_stats.to_csv('./cluster_stats.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"question_stats_columns  = ['question_id',\n #'bundle_id',\n #'part',\n #'kmean_cluster',\n 'bundle_mean_answered_correctly',\n 'bundle_std_answered_correctly',\n 'bundle_perc_students',\n 'bundle_times',\n 'part_mean_answered_correctly',\n 'part_std_answered_correctly',\n 'part_perc_students',\n 'part_times',\n 'cluster_mean_answered_correctly',\n 'cluster_std_answered_correctly',\n 'cluster_perc_students',\n 'cluster_times']\nquestion_stats = questions.to_pandas().merge(bundle_stats, on='bundle_id').merge(part_stats,on = 'part').merge(cluster_stats, on = 'kmean_cluster')\nquestion_stats = question_stats[question_stats_columns]\nquestion_stats.to_csv('./questions_stats_fe.csv', index = False)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del bundle_stats, cluster_stats, part_stats","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}