{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# imports\nimport pandas as pd\nimport numpy as np\nimport nltk\n\n# nltk.download('stopwords')\n\nfrom nltk.corpus import stopwords\nimport string\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import precision_recall_fscore_support, accuracy_score\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom nltk.stem import WordNetLemmatizer\n# Import label encoder\nfrom sklearn import preprocessing\n\nlemmatizer = WordNetLemmatizer()\nw_tokenizer = nltk.tokenize.WhitespaceTokenizer()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T03:13:30.115062Z","iopub.execute_input":"2022-07-05T03:13:30.115840Z","iopub.status.idle":"2022-07-05T03:13:31.339645Z","shell.execute_reply.started":"2022-07-05T03:13:30.115472Z","shell.execute_reply":"2022-07-05T03:13:31.338332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data_train():\n    df_train = pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\n    return df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.342942Z","iopub.execute_input":"2022-07-05T03:13:31.343463Z","iopub.status.idle":"2022-07-05T03:13:31.348574Z","shell.execute_reply.started":"2022-07-05T03:13:31.343413Z","shell.execute_reply":"2022-07-05T03:13:31.347682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lemmatize_text_train(text):\n    return ' '.join([lemmatizer.lemmatize(w) for w in w_tokenizer.tokenize(text)])\n\ndef preprocess_data_train(df_train):\n    stop = set(stopwords.words('english'))\n    \n    # # \n    df_train['discourse_text'] = df_train['discourse_text'].astype(str)\n    df_train['discourse_text'] = df_train['discourse_text'].str.lower()\n    df_train['discourse_text'] = df_train['discourse_text'].apply(lemmatize_text_train)\n    # remove stop words\n    df_train['discourse_text_no_stop'] = df_train['discourse_text'].apply(lambda x: ' '.join([word for word in x.split() if word not in (stop)]))\n\n    for stop_word in ['<br />'] +list(string.punctuation):\n        df_train['discourse_text_no_stop'] = df_train['discourse_text_no_stop'].str.replace(stop_word, '')\n#     df_train['TRANS_CONV_TEXT_no_stop'] = df_train['TRANS_CONV_TEXT_no_stop'].apply(lambda x: x[:10])\n    df_train['discourse_text_no_stop'] = df_train['discourse_type'].str.lower()+' '+df_train['discourse_text_no_stop']\n    \n    cat = pd.DataFrame({'discourse_effectiveness':['Ineffective', 'Adequate', 'Effective'], 'discourse_effectiveness_label': [0, 1,2]})\n    df_train = df_train.merge(cat, how='left', on='discourse_effectiveness')\n    \n    return df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.349851Z","iopub.execute_input":"2022-07-05T03:13:31.351039Z","iopub.status.idle":"2022-07-05T03:13:31.363747Z","shell.execute_reply.started":"2022-07-05T03:13:31.350990Z","shell.execute_reply":"2022-07-05T03:13:31.362541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def vectorize_text_train_train(df_train):\n    vectorizer = CountVectorizer()\n    docs = np.array(df_train['discourse_text_no_stop'])\n    bag = vectorizer.fit_transform(docs)\n    vector = bag.toarray()\n    return vector\n\ndef tfidf_train(df_train):\n    vectorizer = TfidfVectorizer()\n    X = vectorizer.fit_transform(df_train['discourse_text_no_stop'])\n    return vectorizer, X","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.367365Z","iopub.execute_input":"2022-07-05T03:13:31.368111Z","iopub.status.idle":"2022-07-05T03:13:31.375251Z","shell.execute_reply.started":"2022-07-05T03:13:31.368062Z","shell.execute_reply":"2022-07-05T03:13:31.374378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_train_split(x, y):\n    X_train, X_test, y_train, y_test = train_test_split(vector,y, test_size=0.33, random_state=42)\n    return X_train, X_test, y_train, y_test","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.376470Z","iopub.execute_input":"2022-07-05T03:13:31.377233Z","iopub.status.idle":"2022-07-05T03:13:31.392388Z","shell.execute_reply.started":"2022-07-05T03:13:31.377191Z","shell.execute_reply":"2022-07-05T03:13:31.391402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_train(x, y):\n    neigh = KNeighborsClassifier(n_neighbors=9,  weights='uniform', leaf_size=15)\n    neigh.fit(x, y)\n    return neigh","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.393696Z","iopub.execute_input":"2022-07-05T03:13:31.394471Z","iopub.status.idle":"2022-07-05T03:13:31.404123Z","shell.execute_reply.started":"2022-07-05T03:13:31.394437Z","shell.execute_reply":"2022-07-05T03:13:31.403037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test(X_test, y_test, neigh):\n    y_pred = neigh.predict(X_test)\n    return y_pred, precision_recall_fscore_support(y_test, y_pred, average='macro'), accuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.407153Z","iopub.execute_input":"2022-07-05T03:13:31.407898Z","iopub.status.idle":"2022-07-05T03:13:31.416320Z","shell.execute_reply.started":"2022-07-05T03:13:31.407851Z","shell.execute_reply":"2022-07-05T03:13:31.414908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# vectorizer, vector = tfidf(df_input)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.417404Z","iopub.execute_input":"2022-07-05T03:13:31.418066Z","iopub.status.idle":"2022-07-05T03:13:31.426519Z","shell.execute_reply.started":"2022-07-05T03:13:31.418035Z","shell.execute_reply":"2022-07-05T03:13:31.425373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_input = get_data_train()\ndf_input = preprocess_data_train(df_input)\nvectorizer, vector = tfidf_train(df_input)\nX_train, X_test, y_train, y_test = get_test_train_split(vector, df_input['discourse_effectiveness_label'] )\nmodel = get_model_train(X_train, y_train)\ny_pred, f1, acc = test(X_test, y_test, model)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:31.428066Z","iopub.execute_input":"2022-07-05T03:13:31.428785Z","iopub.status.idle":"2022-07-05T03:13:58.737899Z","shell.execute_reply.started":"2022-07-05T03:13:31.428744Z","shell.execute_reply":"2022-07-05T03:13:58.736633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.742556Z","iopub.execute_input":"2022-07-05T03:13:58.743280Z","iopub.status.idle":"2022-07-05T03:13:58.752758Z","shell.execute_reply.started":"2022-07-05T03:13:58.743237Z","shell.execute_reply":"2022-07-05T03:13:58.751489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.754805Z","iopub.execute_input":"2022-07-05T03:13:58.755319Z","iopub.status.idle":"2022-07-05T03:13:58.766539Z","shell.execute_reply.started":"2022-07-05T03:13:58.755256Z","shell.execute_reply":"2022-07-05T03:13:58.765156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"def get_data_predict():\n    df_test = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\", encoding='latin1')\n    return df_test","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.769290Z","iopub.execute_input":"2022-07-05T03:13:58.769728Z","iopub.status.idle":"2022-07-05T03:13:58.778838Z","shell.execute_reply.started":"2022-07-05T03:13:58.769695Z","shell.execute_reply":"2022-07-05T03:13:58.777650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lemmatize_text_predict(text):\n    return ' '.join([lemmatizer.lemmatize(w) for w in w_tokenizer.tokenize(text)])\n\ndef preprocess_data_predict(df_train):\n    stop = set(stopwords.words('english'))\n    \n    # # \n    df_train['discourse_text'] = df_train['discourse_text'].astype(str)\n    df_train['discourse_text'] = df_train['discourse_text'].str.lower()\n    df_train['discourse_text'] = df_train['discourse_text'].apply(lemmatize_text_predict)\n    # remove stop words\n    df_train['discourse_text_no_stop'] = df_train['discourse_text'].apply(lambda x: ' '.join([word for word in x.split() if word not in (stop)]))\n\n    for stop_word in ['<br />'] +list(string.punctuation):\n        df_train['discourse_text_no_stop'] = df_train['discourse_text_no_stop'].str.replace(stop_word, '')\n#     df_train['TRANS_CONV_TEXT_no_stop'] = df_train['TRANS_CONV_TEXT_no_stop'].apply(lambda x: x[:10])\n    df_train['discourse_text_no_stop'] = df_train['discourse_type'].str.lower()+' '+df_train['discourse_text_no_stop']\n    \n#     cat = pd.DataFrame({'discourse_effectiveness':['Ineffective', 'Adequate', 'Effective'], 'discourse_effectiveness_label': [0, 1,2]})\n#     df_train = df_train.merge(cat, how='left', on='discourse_effectiveness')\n    \n    return df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.780247Z","iopub.execute_input":"2022-07-05T03:13:58.780728Z","iopub.status.idle":"2022-07-05T03:13:58.793896Z","shell.execute_reply.started":"2022-07-05T03:13:58.780686Z","shell.execute_reply":"2022-07-05T03:13:58.792865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def vectorize_text_predict(df_train):\n#     vectorizer = CountVectorizer()\n    docs = np.array(df_train['discourse_text_no_stop'])\n    bag = vectorizer.fit_transform(docs)\n    vector = bag.toarray()\n    return vector\n\ndef tfidf_predict(vectorizer, df_train):\n#     vectorizer = TfidfVectorizer()\n    X = vectorizer.transform(df_train['discourse_text_no_stop'])\n    return X","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.795356Z","iopub.execute_input":"2022-07-05T03:13:58.796018Z","iopub.status.idle":"2022-07-05T03:13:58.812000Z","shell.execute_reply.started":"2022-07-05T03:13:58.795983Z","shell.execute_reply":"2022-07-05T03:13:58.810885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_prediction_predict(neigh, vector):\n    pred = neigh.predict(vector)\n    pred = pd.DataFrame(pred)\n    pred = pred.reset_index()\n    pred.columns = ['Index', 'discourse_effectiveness']\n    return pred","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.813599Z","iopub.execute_input":"2022-07-05T03:13:58.814382Z","iopub.status.idle":"2022-07-05T03:13:58.823903Z","shell.execute_reply.started":"2022-07-05T03:13:58.814263Z","shell.execute_reply":"2022-07-05T03:13:58.822860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = get_data_predict()\ntest = preprocess_data_predict(test)\nvector1 =  tfidf_predict(vectorizer, test)\nt_p = get_prediction_predict(model, vector1)\n# t_p.to_csv('output.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.825870Z","iopub.execute_input":"2022-07-05T03:13:58.826790Z","iopub.status.idle":"2022-07-05T03:13:58.885370Z","shell.execute_reply.started":"2022-07-05T03:13:58.826730Z","shell.execute_reply":"2022-07-05T03:13:58.884320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_p = pd.concat([t_p, pd.get_dummies(t_p['discourse_effectiveness'])], axis=1)\nif 0 not in t_p.columns:\n    t_p[0] = 0\nif 1 not in t_p.columns:\n    t_p[1] = 0\nif 2 not in t_p.columns:\n    t_p[2] = 0\nt_p.rename(columns={0: 'Ineffective', 1: 'Adequate', 2: 'Effective'}, inplace=True)\nt_p = pd.concat([t_p, test['discourse_id']], axis=1)\nt_p = t_p[['discourse_id', 'Ineffective', 'Adequate', 'Effective']].copy()\nt_p.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T03:13:58.886658Z","iopub.execute_input":"2022-07-05T03:13:58.887473Z","iopub.status.idle":"2022-07-05T03:13:58.907828Z","shell.execute_reply.started":"2022-07-05T03:13:58.887439Z","shell.execute_reply":"2022-07-05T03:13:58.906865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}