{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sklearn, os\nimport numpy as np\nimport pandas as pd\nimport scipy\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.linear_model import LinearRegression, LogisticRegression\nfrom sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T02:50:48.571930Z","iopub.execute_input":"2022-08-05T02:50:48.572514Z","iopub.status.idle":"2022-08-05T02:50:49.792233Z","shell.execute_reply.started":"2022-08-05T02:50:48.572396Z","shell.execute_reply":"2022-08-05T02:50:49.790889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train files\npath = '/kaggle/input/feedback-prize-effectiveness/'\ntrain_dir = path + 'train/'\ntest_dir = path + 'test/'\n\ntrain_df = pd.read_csv(path + 'train.csv')\ncorpus = train_df['discourse_text']\nlabels = train_df['discourse_effectiveness']\n\ntest_df = pd.read_csv(path + 'test.csv')\ntest_corpus = test_df['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:50:49.794823Z","iopub.execute_input":"2022-08-05T02:50:49.795988Z","iopub.status.idle":"2022-08-05T02:50:50.154507Z","shell.execute_reply.started":"2022-08-05T02:50:49.795940Z","shell.execute_reply":"2022-08-05T02:50:50.153256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(corpus):\n    for i, doc in enumerate(corpus):\n        corpus[i] = doc.lower().rstrip()\n    \n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:50:50.156097Z","iopub.execute_input":"2022-08-05T02:50:50.156572Z","iopub.status.idle":"2022-08-05T02:50:50.163914Z","shell.execute_reply.started":"2022-08-05T02:50:50.156529Z","shell.execute_reply":"2022-08-05T02:50:50.162724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preprocess train files\n# train test split\n\ncorpus = preprocess(corpus)    \ntest_corpus = preprocess(test_corpus)\n\ndic = ['Effective', 'Adequate', 'Ineffective']\nY = []\nfor l in labels:\n    Y.append(dic.index(l))\n\nX_train, X_val, y_train, y_val = train_test_split(corpus, Y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:50:50.167943Z","iopub.execute_input":"2022-08-05T02:50:50.168836Z","iopub.status.idle":"2022-08-05T02:50:59.053387Z","shell.execute_reply.started":"2022-08-05T02:50:50.168791Z","shell.execute_reply":"2022-08-05T02:50:59.052258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nvectorizer = TfidfVectorizer(min_df=150)\nX_t = vectorizer.fit_transform(X_train)\nX_v = vectorizer.transform(X_val)\nX_test = vectorizer.transform(test_corpus)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:50:59.054851Z","iopub.execute_input":"2022-08-05T02:50:59.055359Z","iopub.status.idle":"2022-08-05T02:51:01.800560Z","shell.execute_reply.started":"2022-08-05T02:50:59.055313Z","shell.execute_reply":"2022-08-05T02:51:01.799269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression()\nmode = model.fit(X_t, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:01.802987Z","iopub.execute_input":"2022-08-05T02:51:01.803396Z","iopub.status.idle":"2022-08-05T02:51:02.405523Z","shell.execute_reply.started":"2022-08-05T02:51:01.803358Z","shell.execute_reply":"2022-08-05T02:51:02.404157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred = model.predict(X_t)\nval_pred = model.predict(X_v)\ntest_pred = model.predict(X_test)\n\nkey = np.array([0,1,2])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:02.407380Z","iopub.execute_input":"2022-08-05T02:51:02.408071Z","iopub.status.idle":"2022-08-05T02:51:02.426571Z","shell.execute_reply.started":"2022-08-05T02:51:02.408025Z","shell.execute_reply":"2022-08-05T02:51:02.425262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_logits(preds):\n    key = np.array([0,1,2])\n    out_preds=[]\n    \n    for p in preds:\n        #zrs = [0,0,0]\n        #zrs[np.argmin(abs(key - p))] = 1\n        \n        #out_preds.append(zrs)\n            \n\n        out_preds.append(scipy.special.softmax(-np.abs(p-key)))\n    return out_preds","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:02.428309Z","iopub.execute_input":"2022-08-05T02:51:02.429438Z","iopub.status.idle":"2022-08-05T02:51:02.439526Z","shell.execute_reply.started":"2022-08-05T02:51:02.429387Z","shell.execute_reply":"2022-08-05T02:51:02.438460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame(columns=['discourse_id', 'Ineffective', 'Adequate', 'Effective'])\nkey = np.array([0,1,2])\ntest_preds_logits = to_logits(test_pred)\n\nfor i, row in test_df.iterrows():\n    p = test_preds_logits[i]\n    submit.loc[i] = [row['discourse_id'], p[0], p[1], p[2]]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:02.445565Z","iopub.execute_input":"2022-08-05T02:51:02.446557Z","iopub.status.idle":"2022-08-05T02:51:03.931352Z","shell.execute_reply.started":"2022-08-05T02:51:02.446503Z","shell.execute_reply":"2022-08-05T02:51:03.930151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(mean_squared_error(y_train, train_pred))\nprint(mean_squared_error(y_val, val_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:03.934549Z","iopub.execute_input":"2022-08-05T02:51:03.934927Z","iopub.status.idle":"2022-08-05T02:51:03.946725Z","shell.execute_reply.started":"2022-08-05T02:51:03.934893Z","shell.execute_reply":"2022-08-05T02:51:03.945745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:03.948336Z","iopub.execute_input":"2022-08-05T02:51:03.948947Z","iopub.status.idle":"2022-08-05T02:51:03.958467Z","shell.execute_reply.started":"2022-08-05T02:51:03.948913Z","shell.execute_reply":"2022-08-05T02:51:03.957431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit","metadata":{"execution":{"iopub.status.busy":"2022-08-05T02:51:37.792771Z","iopub.execute_input":"2022-08-05T02:51:37.793355Z","iopub.status.idle":"2022-08-05T02:51:37.817231Z","shell.execute_reply.started":"2022-08-05T02:51:37.793306Z","shell.execute_reply":"2022-08-05T02:51:37.815824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}