{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-28T23:14:03.815620Z","iopub.execute_input":"2023-06-28T23:14:03.816682Z","iopub.status.idle":"2023-06-28T23:14:03.833114Z","shell.execute_reply.started":"2023-06-28T23:14:03.816644Z","shell.execute_reply":"2023-06-28T23:14:03.832259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n=200000\nX = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\",nrows = n)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:03.837914Z","iopub.execute_input":"2023-06-28T23:14:03.838818Z","iopub.status.idle":"2023-06-28T23:14:04.571681Z","shell.execute_reply.started":"2023-06-28T23:14:03.838779Z","shell.execute_reply":"2023-06-28T23:14:04.570611Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\",nrows = n*20)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:04.573266Z","iopub.execute_input":"2023-06-28T23:14:04.573580Z","iopub.status.idle":"2023-06-28T23:14:04.888240Z","shell.execute_reply.started":"2023-06-28T23:14:04.573554Z","shell.execute_reply":"2023-06-28T23:14:04.886898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nXs = X# for kaggle\nc_vec = CountVectorizer(ngram_range=(2,8), max_features = 100)\n\n# input to fit_transform() should be an iterable with strings\nngrams = c_vec.fit_transform(Xs[\"text\"].dropna(how = \"any\", axis=0).values)\n\nvocab = c_vec.vocabulary_\n\ncount_values = ngrams.toarray().sum(axis=0)#%automagic\n\nprint(count_values[0:20])","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:04.891619Z","iopub.execute_input":"2023-06-28T23:14:04.891981Z","iopub.status.idle":"2023-06-28T23:14:07.727820Z","shell.execute_reply.started":"2023-06-28T23:14:04.891951Z","shell.execute_reply":"2023-06-28T23:14:07.726673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ng_count, ng_text in sorted([(count_values[i],k) for k,i in vocab.items()][0:20], reverse=True):\n    print(ng_count, ng_text)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:07.729351Z","iopub.execute_input":"2023-06-28T23:14:07.729678Z","iopub.status.idle":"2023-06-28T23:14:07.736735Z","shell.execute_reply.started":"2023-06-28T23:14:07.729650Z","shell.execute_reply":"2023-06-28T23:14:07.735577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ngrams = np.asarray(ngrams.todense())\nfrom sklearn.decomposition import PCA\npca = PCA(n_components=20)\nngrams = pca.fit_transform(ngrams)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:07.738370Z","iopub.execute_input":"2023-06-28T23:14:07.738746Z","iopub.status.idle":"2023-06-28T23:14:08.676427Z","shell.execute_reply.started":"2023-06-28T23:14:07.738715Z","shell.execute_reply":"2023-06-28T23:14:08.675429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[\"text\"] = X[\"text\"].fillna(\"\")\nXs = pca.transform(np.asarray(c_vec.transform(X[\"text\"]).todense()))\nXs = pd.DataFrame(Xs, columns=[str(i) for i in range(Xs.shape[1])])\nX_vectorized = pd.concat([X, Xs],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:08.681716Z","iopub.execute_input":"2023-06-28T23:14:08.684619Z","iopub.status.idle":"2023-06-28T23:14:11.789390Z","shell.execute_reply.started":"2023-06-28T23:14:08.684580Z","shell.execute_reply":"2023-06-28T23:14:11.788306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:11.791097Z","iopub.execute_input":"2023-06-28T23:14:11.791813Z","iopub.status.idle":"2023-06-28T23:14:11.950201Z","shell.execute_reply.started":"2023-06-28T23:14:11.791765Z","shell.execute_reply":"2023-06-28T23:14:11.949341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom time import time\n\nsummary_functions = {\"min\": lambda x: min(x) if len(x)>0 else 0.0, \"max\": lambda x: max(x) if len(x)>0 else 0.0, \n                     \"mean\" : lambda x: np.mean(x) if len(x) > 0 else 0.0}\nsummary_features = [str(i) for i in range(20)][0:5]\nsummary_labels = summary_functions.keys()\ncolumns = []\nfor j in summary_features:\n    for k in summary_labels:\n                     columns.append(j + \"_\" + k)\ndf2 = {i: []  for i in columns}\n\ndf =X_vectorized\n\ndf.sort_values(by = \"session_id\")\n\nt0 = time()\n\n#from numba import jit\n\n#@jit\ndef groupReshape(df, summary_functions, summary_features):\n    summary_labels = summary_functions.keys()\n    columns = []\n    for j in summary_features:\n        for k in summary_labels:\n                         columns.append(j + \"_\" + k)\n    df2 = {i: []  for i in columns}\n\n    p = 0 # dataframestarter\n    p_start = p\n    #uniq_ids and df must be sorted in the same order\n    l = len(df)\n\n    uniq_id =  df.iloc[0][\"session_id\"]\n    uniq_ids = [uniq_id,]\n\n    while p<l:\n\n        id_new =  df.iloc[p][\"session_id\"]\n        if uniq_id == id_new:\n            p+=1\n\n            continue\n\n        else:\n\n            uniq_ids.append(id_new)\n            uniq_id = id_new\n            #print(id_new)\n            p+=1\n\n\n        for j in summary_features:\n            s = df.iloc[p_start:p][j].values\n            #print(s)\n            for k in summary_labels:\n                #print(k)\n                df2[j + \"_\" + k].append(summary_functions[k](s))\n            #del s\n            #gc.collect()\n\n        p_start = p\n    #last one will have p>l\n    for j in summary_features:\n        s = df.iloc[p_start:p][j].values\n        #print(s)\n        for k in summary_labels:\n            #print(k)\n            df2[j + \"_\" + k].append(summary_functions[k](s))\n            #del s\n    df2[\"session_id\"] = uniq_ids\n    return df2\n\n\n\nt0 = time()\n\ndf2 = groupReshape(df, summary_functions, summary_features)\n\n\nprint(time()-t0)  ","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-28T23:14:11.951681Z","iopub.execute_input":"2023-06-28T23:14:11.952346Z","iopub.status.idle":"2023-06-28T23:14:42.139074Z","shell.execute_reply.started":"2023-06-28T23:14:11.952311Z","shell.execute_reply":"2023-06-28T23:14:42.138114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped =  pd.DataFrame.from_dict(df2)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:42.140576Z","iopub.execute_input":"2023-06-28T23:14:42.141242Z","iopub.status.idle":"2023-06-28T23:14:42.149079Z","shell.execute_reply.started":"2023-06-28T23:14:42.141208Z","shell.execute_reply":"2023-06-28T23:14:42.148138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=y\ndf[['session_id', 'question']] = pd.DataFrame(map(lambda x : str(x).split('_q',),y['session_id'].values),columns = [\"session\", \"question\"])\nprint(df.tail)\nwide_data = df.pivot(index='session_id', columns='question', values='correct')\nwide_data = wide_data.reset_index()\n\nwide_data[\"session_id\"] = wide_data[\"session_id\"].astype(\"str\")\nwide_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:42.150231Z","iopub.execute_input":"2023-06-28T23:14:42.150900Z","iopub.status.idle":"2023-06-28T23:14:43.391039Z","shell.execute_reply.started":"2023-06-28T23:14:42.150864Z","shell.execute_reply":"2023-06-28T23:14:43.389771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ngrouped[\"session_id\"] = grouped[\"session_id\"].astype(\"str\")\ngrouped_merged = grouped.merge(wide_data, on = \"session_id\", how = \"left\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:43.392589Z","iopub.execute_input":"2023-06-28T23:14:43.392923Z","iopub.status.idle":"2023-06-28T23:14:43.413368Z","shell.execute_reply.started":"2023-06-28T23:14:43.392896Z","shell.execute_reply":"2023-06-28T23:14:43.412241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_merged.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:43.419803Z","iopub.execute_input":"2023-06-28T23:14:43.420138Z","iopub.status.idle":"2023-06-28T23:14:43.427868Z","shell.execute_reply.started":"2023-06-28T23:14:43.420111Z","shell.execute_reply":"2023-06-28T23:14:43.426854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:43.429048Z","iopub.execute_input":"2023-06-28T23:14:43.429387Z","iopub.status.idle":"2023-06-28T23:14:43.442306Z","shell.execute_reply.started":"2023-06-28T23:14:43.429359Z","shell.execute_reply":"2023-06-28T23:14:43.441385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"modelList = []\nfrom sklearn.ensemble import GradientBoostingClassifier\n\n#edit this after selection\nmodelColumnList = []\n\nfor i in range(18):\n    # model = CatBoostClassifier(iterations=1000,\n    #                        task_type=\"GPU\")\n    X = grouped_merged[columns]\n    #X = X.drop(\"session_id\", axis = 1)\n    \n    y = grouped_merged[str(i+1)]\n    \n    #modelList.append(GradientBoostingClassifier(**modelColumnList[i]).fit(X, y))\n    modelList.append(GradientBoostingClassifier().fit(X, y))\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:43.443336Z","iopub.execute_input":"2023-06-28T23:14:43.444313Z","iopub.status.idle":"2023-06-28T23:14:45.741337Z","shell.execute_reply.started":"2023-06-28T23:14:43.444276Z","shell.execute_reply":"2023-06-28T23:14:45.740367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:45.742507Z","iopub.execute_input":"2023-06-28T23:14:45.743511Z","iopub.status.idle":"2023-06-28T23:14:45.768782Z","shell.execute_reply.started":"2023-06-28T23:14:45.743475Z","shell.execute_reply":"2023-06-28T23:14:45.767999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder_310 as jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()\n\ncounter = 0\n# The API will deliver two dataframes in this specific order,\n# for every session+level grouping (one group per session for each checkpoint)\nfor (test, sample_submission) in iter_test:\n    if counter == 0:\n        print(sample_submission.head())\n        #print(test.columns)\n    #print(test.shape)\n    \n    \n    \n    \n    \n    \n    \n    \n    \n    \n    \n    test[\"text\"] = test[\"text\"].fillna(\"\")\n    tests = pca.transform(np.asarray(c_vec.transform(test[\"text\"]).todense()))\n    tests = pd.DataFrame(tests, columns=[str(i) for i in range(Xs.shape[1])])\n    test_vectorized = pd.concat([test, tests],axis = 1)\n\n\n\n    from time import time\n\n    summary_functions = {\"min\": lambda x: min(x) if len(x)>0 else 0.0, \"max\": lambda x: max(x) if len(x)>0 else 0.0, \n                         \"mean\" : lambda x: np.mean(x) if len(x) > 0 else 0.0}\n    summary_features = [str(i) for i in range(20)][0:5]\n    summary_labels = summary_functions.keys()\n    columns = []\n    for j in summary_features:\n        for k in summary_labels:\n                         columns.append(j + \"_\" + k)\n    df2 = {i: []  for i in columns}\n\n    df0 =test_vectorized\n\n    df0.sort_values(by = \"session_id\")\n\n    t0 = time()\n\n    #from numba import jit\n\n    #@jit\n    def groupReshape(df, summary_functions, summary_features):\n        summary_labels = summary_functions.keys()\n        columns = []\n        for j in summary_features:\n            for k in summary_labels:\n                             columns.append(j + \"_\" + k)\n        df2 = {i: []  for i in columns}\n\n        p = 0 # dataframestarter\n        p_start = p\n        #uniq_ids and df must be sorted in the same order\n        l = len(df)\n\n        uniq_id =  df.iloc[0][\"session_id\"]\n        uniq_ids = [uniq_id,]\n\n        while p<l:\n\n            id_new =  df.iloc[p][\"session_id\"]\n            if uniq_id == id_new:\n                p+=1\n\n                continue\n\n            else:\n\n                uniq_ids.append(id_new)\n                uniq_id = id_new\n                print(id_new)\n                p+=1\n\n\n            for j in summary_features:\n                s = df.iloc[p_start:p][j].values\n                #print(s)\n                for k in summary_labels:\n                    #print(k)\n                    df2[j + \"_\" + k].append(summary_functions[k](s))\n                #del s\n                #gc.collect()\n\n            p_start = p\n        #last one will have p>l\n        for j in summary_features:\n            s = df.iloc[p_start:p][j].values\n            #print(s)\n            for k in summary_labels:\n                #print(k)\n                df2[j + \"_\" + k].append(summary_functions[k](s))\n                #del s\n        df2[\"session_id\"] = uniq_ids\n        return df2\n\n\n\n    t0 = time()\n\n    df2 = groupReshape(df0, summary_functions, summary_features)\n    df2 =  pd.DataFrame.from_dict(df2)\n    n = df2.shape[1]\n\n    print(time()-t0)  \n    \n    \n    ## users make predictions here using the test data\n    sample_submission['question'] = [int(label.split('_')[1][1:]) for label in sample_submission['session_id']]\n    df = sample_submission\n    for i in range(18):\n        df.loc[df.question == 1+i, 'correct'] = modelList[i].predict(df2.drop(\"session_id\", axis=1))\n    df[\"level\"] = min(test.loc[test['session_id'] == df2[\"session_id\"][0] ]['level'])\n\n\n    ## env.predict appends the session+level sample_submission to the overall\n    ## submission\n\n    env.predict(sample_submission[['session_id', 'correct','level']])\n    print(sample_submission)\n    counter += 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T23:14:45.770351Z","iopub.execute_input":"2023-06-28T23:14:45.770894Z","iopub.status.idle":"2023-06-28T23:14:47.056374Z","shell.execute_reply.started":"2023-06-28T23:14:45.770862Z","shell.execute_reply":"2023-06-28T23:14:47.055338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}