{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!ln -s /kaggle/input/omw-1-4 /usr/share/nltk_data/corpora/omw-1.4 \n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:51.042230Z","iopub.execute_input":"2022-08-14T23:42:51.042991Z","iopub.status.idle":"2022-08-14T23:42:52.214514Z","shell.execute_reply.started":"2022-08-14T23:42:51.042862Z","shell.execute_reply":"2022-08-14T23:42:52.213154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport time\nimport numpy as np\nimport pandas as pd\n\nimport string\nimport nltk\nfrom nltk.tokenize import word_tokenize, sent_tokenize\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.corpus import wordnet, stopwords\n\nfrom sklearn import pipeline as pip\nfrom sklearn.feature_extraction import text\nfrom sklearn import preprocessing as pre\nfrom sklearn import decomposition as dec\nfrom sklearn import ensemble as ens\nfrom sklearn import compose as com\n\nfrom sklearn.base import BaseEstimator, TransformerMixin","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:52.216836Z","iopub.execute_input":"2022-08-14T23:42:52.218028Z","iopub.status.idle":"2022-08-14T23:42:54.363137Z","shell.execute_reply.started":"2022-08-14T23:42:52.217989Z","shell.execute_reply":"2022-08-14T23:42:54.362003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# util funcs\ndef simple_tokenizer(text):\n    \"\"\"\n    1. decode text into utf-8 format\n    2. tokenize the text to words\n    3. remove punctuations\n    4. Lemmatize the items \n    \n    \"\"\"\n    text = text.encode('ascii', errors='ignore').decode(\"utf-8\")\n    text = \"\".join([c for c in text if c not in string.punctuation])\n    \n    tokens = nltk.word_tokenize(text)\n    tokens = [token for token in tokens if len(token) > 3 and len(token) <= 11]\n    \n    \n    \n    wnl = WordNetLemmatizer()\n    return [wnl.lemmatize(item) for item in tokens]\n    \nclass WordLengthCreator(BaseEstimator, TransformerMixin):\n    \n    def fit(self, X, y=None):\n        return self\n\n    def transform(self, X, y=None):    \n        text_len = X.str.len()\n        token_len = X.apply(lambda x: len(x.split()))\n        avg_word_len = (text_len / token_len).values.reshape(-1, 1)\n        \n        return avg_word_len ","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:54.364507Z","iopub.execute_input":"2022-08-14T23:42:54.364947Z","iopub.status.idle":"2022-08-14T23:42:54.375439Z","shell.execute_reply.started":"2022-08-14T23:42:54.364912Z","shell.execute_reply":"2022-08-14T23:42:54.374098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_PATH = \"../input/feedback-prize-effectiveness/\"\nTRAIN_SET = os.path.join(DATASET_PATH, \"train.csv\")\nTEST_SET = os.path.join(DATASET_PATH, \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:54.378409Z","iopub.execute_input":"2022-08-14T23:42:54.379986Z","iopub.status.idle":"2022-08-14T23:42:54.388782Z","shell.execute_reply.started":"2022-08-14T23:42:54.379947Z","shell.execute_reply":"2022-08-14T23:42:54.387348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_SET)\ntest = pd.read_csv(TEST_SET)\n\nprint(r\"Train Set: \", len(train))\nprint(r\"Test Set: \", len(test))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:54.390713Z","iopub.execute_input":"2022-08-14T23:42:54.391164Z","iopub.status.idle":"2022-08-14T23:42:54.707317Z","shell.execute_reply.started":"2022-08-14T23:42:54.391107Z","shell.execute_reply":"2022-08-14T23:42:54.706000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train[[\"discourse_text\", \"discourse_type\"]]\ny_train = train[\"discourse_effectiveness\"]\ny_train = y_train.map({'Ineffective': 1, 'Effective': 2, 'Adequate': 3}, -1)\n\nX_test = test[[\"discourse_text\", \"discourse_type\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:54.708904Z","iopub.execute_input":"2022-08-14T23:42:54.709297Z","iopub.status.idle":"2022-08-14T23:42:54.739132Z","shell.execute_reply.started":"2022-08-14T23:42:54.709242Z","shell.execute_reply":"2022-08-14T23:42:54.738006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Text Transformer_pipe\npipe_transformer = pip.Pipeline([\n    (\"tfidf\", text.TfidfVectorizer(lowercase=True, ngram_range=(1,1), tokenizer=simple_tokenizer, strip_accents=\"unicode\", smooth_idf=True)),\n    (\"scaler\", pre.StandardScaler(with_mean=False)),\n    (\"svd\", dec.TruncatedSVD(n_components=100, random_state=42, n_iter=10)),\n])\n\n# Merge Transformers\nmyTransformer = com.ColumnTransformer([\n    (\"discourse_text_transformer\", pipe_transformer, \"discourse_text\"),\n    (\"word_length_transformer\", WordLengthCreator(), \"discourse_text\"),\n    (\"effectiveness_transformer\", pre.OneHotEncoder(drop=\"first\", handle_unknown=\"ignore\"), [\"discourse_type\"])\n], remainder=\"passthrough\", n_jobs=-1)\n\n# Create Model Pipeline\nrf_pipeline = pip.Pipeline([\n    (\"all_transformers\", myTransformer),\n    (\"clf\", ens.RandomForestClassifier(random_state=42, n_jobs=-1,  class_weight=None,\n                                       criterion='entropy',\n                                       max_depth=15,\n                                       max_features=0.5, \n                                       min_samples_leaf=9,\n                                       min_samples_split=10,\n                                       n_estimators=250)),\n])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:54.740404Z","iopub.execute_input":"2022-08-14T23:42:54.741302Z","iopub.status.idle":"2022-08-14T23:42:54.752988Z","shell.execute_reply.started":"2022-08-14T23:42:54.741267Z","shell.execute_reply":"2022-08-14T23:42:54.751445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit and Predict\nrf_pipeline.fit(X_train, y_train)\n\ny_pred_test = rf_pipeline.predict_proba(X_test)\ncolumns = [\"Ineffective\", \"Effective\", \"Adequate\"]\n\ndf_result = pd.DataFrame(y_pred_test, columns=columns)\ndf_result[\"discourse_id\"] = test.discourse_id","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:42:54.754544Z","iopub.execute_input":"2022-08-14T23:42:54.758620Z","iopub.status.idle":"2022-08-14T23:47:48.300391Z","shell.execute_reply.started":"2022-08-14T23:42:54.758583Z","shell.execute_reply":"2022-08-14T23:47:48.299285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission\ndf_result[[\"discourse_id\", \"Ineffective\", \"Adequate\", \"Effective\"]].to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T23:47:48.304887Z","iopub.execute_input":"2022-08-14T23:47:48.305234Z","iopub.status.idle":"2022-08-14T23:47:48.317127Z","shell.execute_reply.started":"2022-08-14T23:47:48.305203Z","shell.execute_reply":"2022-08-14T23:47:48.316155Z"},"trusted":true},"execution_count":null,"outputs":[]}]}