{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T10:53:59.293798Z","iopub.execute_input":"2022-08-14T10:53:59.294224Z","iopub.status.idle":"2022-08-14T10:54:00.678839Z","shell.execute_reply.started":"2022-08-14T10:53:59.294188Z","shell.execute_reply":"2022-08-14T10:54:00.677807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom scipy import sparse\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:00.680895Z","iopub.execute_input":"2022-08-14T10:54:00.681250Z","iopub.status.idle":"2022-08-14T10:54:01.512933Z","shell.execute_reply.started":"2022-08-14T10:54:00.681217Z","shell.execute_reply":"2022-08-14T10:54:01.511480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train= pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:01.515260Z","iopub.execute_input":"2022-08-14T10:54:01.516023Z","iopub.status.idle":"2022-08-14T10:54:01.874531Z","shell.execute_reply.started":"2022-08-14T10:54:01.515974Z","shell.execute_reply":"2022-08-14T10:54:01.873351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test= pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:01.876936Z","iopub.execute_input":"2022-08-14T10:54:01.877363Z","iopub.status.idle":"2022-08-14T10:54:01.894058Z","shell.execute_reply.started":"2022-08-14T10:54:01.877332Z","shell.execute_reply":"2022-08-14T10:54:01.892789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text']= train['essay_id'].apply(lambda x: open(f'../input/feedback-prize-effectiveness/train/{x}.txt').read())\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:01.895525Z","iopub.execute_input":"2022-08-14T10:54:01.896337Z","iopub.status.idle":"2022-08-14T10:54:29.788601Z","shell.execute_reply.started":"2022-08-14T10:54:01.896302Z","shell.execute_reply":"2022-08-14T10:54:29.787156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['text']= test['essay_id'].apply(lambda x: open(f'../input/feedback-prize-effectiveness/test/{x}.txt').read())\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:29.790013Z","iopub.execute_input":"2022-08-14T10:54:29.790428Z","iopub.status.idle":"2022-08-14T10:54:29.814781Z","shell.execute_reply.started":"2022-08-14T10:54:29.790396Z","shell.execute_reply":"2022-08-14T10:54:29.813674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"effectiveness_map= {\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2}\ntrain['target']= train[\"discourse_effectiveness\"].map(effectiveness_map)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:29.817303Z","iopub.execute_input":"2022-08-14T10:54:29.818143Z","iopub.status.idle":"2022-08-14T10:54:29.834818Z","shell.execute_reply.started":"2022-08-14T10:54:29.818103Z","shell.execute_reply":"2022-08-14T10:54:29.833614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train= train.reset_index(drop=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:29.836602Z","iopub.execute_input":"2022-08-14T10:54:29.837704Z","iopub.status.idle":"2022-08-14T10:54:29.869661Z","shell.execute_reply.started":"2022-08-14T10:54:29.837666Z","shell.execute_reply":"2022-08-14T10:54:29.868821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import log_loss\n\npreds=[]\n\ntrain_ = train #use all the discourse_ids which are not marked by current fold index\neval_ = train #use current fold index rows as validation set\n         \n# Training, Validation, and Test Dataset\n#discourse_id\ntf = TfidfVectorizer(ngram_range=(1,2),norm='l2', smooth_idf=True)\ntrain_discourse_tfidf = tf.fit_transform(train_[\"discourse_text\"])\neval_discourse_tfidf = tf.transform(eval_[\"discourse_text\"])\ntest_discourse_tfidf = tf.transform(test[\"discourse_text\"])\n\n\n#text\ntf = TfidfVectorizer(ngram_range=(1,2),norm='l2', smooth_idf=True) # Load tf another time because it will learn the new vocabulary for 'text'\ntrain_text_tfidf = tf.fit_transform(train_[\"text\"])\neval_text_tfidf = tf.transform(eval_[\"text\"])\ntest_text_tfidf = tf.transform(test[\"text\"])\n\n#discourse_type\nohe = OneHotEncoder()\ntrain_type_ohe =  sparse.csr_matrix(ohe.fit_transform(train_[\"discourse_type\"].values.reshape(-1,1)))\neval_type_ohe =  sparse.csr_matrix(ohe.transform(eval_[\"discourse_type\"].values.reshape(-1,1)))\ntest_type_ohe =  sparse.csr_matrix(ohe.transform(test[\"discourse_type\"].values.reshape(-1,1)))\n\n#Stack each vector representations \ntrain_tfidf = sparse.hstack((train_type_ohe,train_discourse_tfidf,train_text_tfidf))\neval_tfidf = sparse.hstack((eval_type_ohe,eval_discourse_tfidf,eval_text_tfidf))\ntest_tfidf = sparse.hstack((test_type_ohe,test_discourse_tfidf,test_text_tfidf))\n\n#Model\nclf = RandomForestClassifier()\nclf.fit(train_tfidf, train_[\"target\"].values)\n\n#Validation \nev_preds = clf.predict_proba(eval_tfidf)\nev_loss = log_loss(eval_[\"target\"].values,ev_preds)\n\n\n#Test\npreds.append(clf.predict_proba(test_tfidf))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T10:54:29.870946Z","iopub.execute_input":"2022-08-14T10:54:29.871836Z","iopub.status.idle":"2022-08-14T11:22:02.747963Z","shell.execute_reply.started":"2022-08-14T10:54:29.871802Z","shell.execute_reply":"2022-08-14T11:22:02.746643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T11:22:02.751447Z","iopub.execute_input":"2022-08-14T11:22:02.751817Z","iopub.status.idle":"2022-08-14T11:22:02.764501Z","shell.execute_reply.started":"2022-08-14T11:22:02.751783Z","shell.execute_reply":"2022-08-14T11:22:02.763163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_preds= np.array(preds).mean(0)\nprint(full_preds.shape)\nsubmission.loc[:, \"Ineffective\"]= full_preds[:,0]\nsubmission.loc[:, \"Adequate\"]= full_preds[:,1]\nsubmission.loc[:, \"Effective\"]= full_preds[:,2]\nsubmission.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T11:22:02.766209Z","iopub.execute_input":"2022-08-14T11:22:02.766921Z","iopub.status.idle":"2022-08-14T11:22:02.786796Z","shell.execute_reply.started":"2022-08-14T11:22:02.766885Z","shell.execute_reply":"2022-08-14T11:22:02.785663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T11:22:02.788507Z","iopub.execute_input":"2022-08-14T11:22:02.789821Z","iopub.status.idle":"2022-08-14T11:22:02.799309Z","shell.execute_reply.started":"2022-08-14T11:22:02.789776Z","shell.execute_reply":"2022-08-14T11:22:02.797971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}