{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport re\nimport string\nimport nltk\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nstop_words = stopwords.words('english')\nnltk.download('wordnet')\nfrom nltk.stem.wordnet import WordNetLemmatizer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T10:52:17.053112Z","iopub.execute_input":"2022-07-19T10:52:17.053614Z","iopub.status.idle":"2022-07-19T10:52:19.202831Z","shell.execute_reply.started":"2022-07-19T10:52:17.053496Z","shell.execute_reply":"2022-07-19T10:52:19.201030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/nlp-getting-started/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:52:31.093031Z","iopub.execute_input":"2022-07-19T10:52:31.094395Z","iopub.status.idle":"2022-07-19T10:52:31.146747Z","shell.execute_reply.started":"2022-07-19T10:52:31.094353Z","shell.execute_reply":"2022-07-19T10:52:31.145341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:52:37.141906Z","iopub.execute_input":"2022-07-19T10:52:37.142360Z","iopub.status.idle":"2022-07-19T10:52:37.170468Z","shell.execute_reply.started":"2022-07-19T10:52:37.142321Z","shell.execute_reply":"2022-07-19T10:52:37.169593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:52:43.243653Z","iopub.execute_input":"2022-07-19T10:52:43.244072Z","iopub.status.idle":"2022-07-19T10:52:43.251610Z","shell.execute_reply.started":"2022-07-19T10:52:43.244040Z","shell.execute_reply":"2022-07-19T10:52:43.250332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:52:51.100903Z","iopub.execute_input":"2022-07-19T10:52:51.101584Z","iopub.status.idle":"2022-07-19T10:52:51.113557Z","shell.execute_reply.started":"2022-07-19T10:52:51.101542Z","shell.execute_reply":"2022-07-19T10:52:51.112572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=train['target'], hue=train['target'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:52:57.492864Z","iopub.execute_input":"2022-07-19T10:52:57.493273Z","iopub.status.idle":"2022-07-19T10:52:57.684174Z","shell.execute_reply.started":"2022-07-19T10:52:57.493242Z","shell.execute_reply":"2022-07-19T10:52:57.683066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:04.141084Z","iopub.execute_input":"2022-07-19T10:53:04.141542Z","iopub.status.idle":"2022-07-19T10:53:04.153154Z","shell.execute_reply.started":"2022-07-19T10:53:04.141505Z","shell.execute_reply":"2022-07-19T10:53:04.151937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(50):\n    print(train['text'][i])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:08.790676Z","iopub.execute_input":"2022-07-19T10:53:08.791130Z","iopub.status.idle":"2022-07-19T10:53:08.799182Z","shell.execute_reply.started":"2022-07-19T10:53:08.791089Z","shell.execute_reply":"2022-07-19T10:53:08.797787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_URL(text):\n    url = re.compile(r'https?://\\S+')\n    return url.sub(r' httpsmark ', text)\n\n\ndef remove_html(text):\n    html = re.compile(r'<.*?>')\n    return html.sub(r'', text)\n\n\ndef remove_atsymbol(text):\n    name = re.compile(r'@\\S+')\n    return name.sub(r' atsymbol ', text)\n\n\ndef remove_hashtag(text):\n    hashtag = re.compile(r'#')\n    return hashtag.sub(r' hashtag ', text)\n\n\ndef remove_exclamation(text):\n    exclamation = re.compile(r'!')\n    return exclamation.sub(r' exclamation ', text)\n\n\ndef remove_question(text):\n    question = re.compile(r'?')\n    return question.sub(r' question ', text)\n\n\ndef remove_punc(text):\n    return text.translate(str.maketrans('','',string.punctuation))\n\n\ndef remove_number(text):\n    number = re.compile(r'\\d+')\n    return number.sub(r' number ', text)\n\n\ndef remove_emoji(string):\n    emoji_pattern = re.compile(\"[\"\n                               u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                               u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                               u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                               u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                               u\"\\U00002500-\\U00002BEF\"  # chinese char\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U000024C2-\\U0001F251\"\n                               u\"\\U0001f926-\\U0001f937\"\n                               u\"\\U00010000-\\U0010ffff\"\n                               u\"\\u2640-\\u2642\"\n                               u\"\\u2600-\\u2B55\"\n                               u\"\\u200d\"\n                               u\"\\u23cf\"\n                               u\"\\u23e9\"\n                               u\"\\u231a\"\n                               u\"\\ufe0f\"  # dingbats\n                               u\"\\u3030\"\n                               \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r' emoji ', string)\n\ntrain['text'] = train['text'].str.lower()\ntrain['text'] = train['text'].apply(lambda text: remove_URL(text))\ntrain['text'] = train['text'].apply(lambda text: remove_html(text))\ntrain['text'] = train['text'].apply(lambda text: remove_atsymbol(text))\ntrain['text'] = train['text'].apply(lambda text: remove_hashtag(text))\ntrain['text'] = train['text'].apply(lambda text: remove_exclamation(text))\ntrain['text'] = train['text'].apply(lambda text: remove_punc(text))\ntrain['text'] = train['text'].apply(lambda text: remove_number(text))\ntrain['text'] = train['text'].apply(lambda text: remove_emoji(text))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:17.309693Z","iopub.execute_input":"2022-07-19T10:53:17.310200Z","iopub.status.idle":"2022-07-19T10:53:17.675491Z","shell.execute_reply.started":"2022-07-19T10:53:17.310157Z","shell.execute_reply":"2022-07-19T10:53:17.674198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:25.500652Z","iopub.execute_input":"2022-07-19T10:53:25.501079Z","iopub.status.idle":"2022-07-19T10:53:25.521938Z","shell.execute_reply.started":"2022-07-19T10:53:25.501042Z","shell.execute_reply":"2022-07-19T10:53:25.520900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(train.shape[0]):\n    words = word_tokenize(train['text'][i])\n    sentence = ' '.join([word for word in words if word not in stop_words])\n    train['text'][i] = sentence","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:33.729721Z","iopub.execute_input":"2022-07-19T10:53:33.730150Z","iopub.status.idle":"2022-07-19T10:53:38.836858Z","shell.execute_reply.started":"2022-07-19T10:53:33.730116Z","shell.execute_reply":"2022-07-19T10:53:38.835815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer = nltk.stem.PorterStemmer()\nfor i in range(train.shape[0]):\n    words = word_tokenize(train['text'][i])\n    sentence = ' '.join([stemmer.stem(word) for word in words if word not in stop_words])\n    train['text'][i] = sentence","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:43.498196Z","iopub.execute_input":"2022-07-19T10:53:43.498596Z","iopub.status.idle":"2022-07-19T10:53:52.013854Z","shell.execute_reply.started":"2022-07-19T10:53:43.498565Z","shell.execute_reply":"2022-07-19T10:53:52.012719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmatizer = WordNetLemmatizer()\nfor i in range(train.shape[0]):\n    words = word_tokenize(train['text'][i])\n    sentence = ' '.join([lemmatizer.lemmatize(word) for word in words if word not in stop_words])\n    train['text'][i] = sentence","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:52.016076Z","iopub.execute_input":"2022-07-19T10:53:52.016440Z","iopub.status.idle":"2022-07-19T10:54:00.339436Z","shell.execute_reply.started":"2022-07-19T10:53:52.016407Z","shell.execute_reply":"2022-07-19T10:54:00.338600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nvectorizer = CountVectorizer()\nx_train = vectorizer.fit_transform(train['text'])\ny_train=train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:00.341033Z","iopub.execute_input":"2022-07-19T10:54:00.341736Z","iopub.status.idle":"2022-07-19T10:54:00.535511Z","shell.execute_reply.started":"2022-07-19T10:54:00.341702Z","shell.execute_reply":"2022-07-19T10:54:00.534619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.linear_model import RidgeClassifier\nfrom sklearn.svm import LinearSVC\n\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:01.408070Z","iopub.execute_input":"2022-07-19T10:54:01.408849Z","iopub.status.idle":"2022-07-19T10:54:01.661647Z","shell.execute_reply.started":"2022-07-19T10:54:01.408807Z","shell.execute_reply":"2022-07-19T10:54:01.660524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rc =RidgeClassifier()\nmodel0=rc.fit(x_train, y_train)\nprint(\"train accuracy:\",model0.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:06.450966Z","iopub.execute_input":"2022-07-19T10:54:06.451380Z","iopub.status.idle":"2022-07-19T10:54:06.567517Z","shell.execute_reply.started":"2022-07-19T10:54:06.451348Z","shell.execute_reply":"2022-07-19T10:54:06.566569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#logistic regression\nlr = LogisticRegression(max_iter=2000,penalty='l2')\nmodel1=lr.fit(x_train, y_train)\nprint(\"train accuracy:\",model1.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:11.290109Z","iopub.execute_input":"2022-07-19T10:54:11.290500Z","iopub.status.idle":"2022-07-19T10:54:11.718738Z","shell.execute_reply.started":"2022-07-19T10:54:11.290467Z","shell.execute_reply":"2022-07-19T10:54:11.717718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svm =LinearSVC()\nmodel2=svm.fit(x_train, y_train)\nprint(\"train accuracy:\",model2.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:17.351502Z","iopub.execute_input":"2022-07-19T10:54:17.351894Z","iopub.status.idle":"2022-07-19T10:54:17.825274Z","shell.execute_reply.started":"2022-07-19T10:54:17.351858Z","shell.execute_reply":"2022-07-19T10:54:17.824142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt=DecisionTreeClassifier()\nmodel3=dt.fit(x_train, y_train)\nprint(\"train accuracy:\",model3.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:22.462167Z","iopub.execute_input":"2022-07-19T10:54:22.462593Z","iopub.status.idle":"2022-07-19T10:54:24.164216Z","shell.execute_reply.started":"2022-07-19T10:54:22.462557Z","shell.execute_reply":"2022-07-19T10:54:24.163071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf=RandomForestClassifier(random_state=1234)\nmodel4=rf.fit(x_train, y_train)\nprint(\"train accuracy:\",model4.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:27.730677Z","iopub.execute_input":"2022-07-19T10:54:27.731150Z","iopub.status.idle":"2022-07-19T10:54:41.991813Z","shell.execute_reply.started":"2022-07-19T10:54:27.731109Z","shell.execute_reply":"2022-07-19T10:54:41.990497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbm=GradientBoostingClassifier()\nmodel5=gbm.fit(x_train, y_train)\nprint(\"train accuracy:\",model5.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:41.993745Z","iopub.execute_input":"2022-07-19T10:54:41.994226Z","iopub.status.idle":"2022-07-19T10:54:44.520536Z","shell.execute_reply.started":"2022-07-19T10:54:41.994188Z","shell.execute_reply":"2022-07-19T10:54:44.519440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ada=AdaBoostClassifier()\nmodel6=ada.fit(x_train, y_train)\nprint(\"train accuracy:\",model6.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:44.521923Z","iopub.execute_input":"2022-07-19T10:54:44.522331Z","iopub.status.idle":"2022-07-19T10:54:45.310083Z","shell.execute_reply.started":"2022-07-19T10:54:44.522299Z","shell.execute_reply":"2022-07-19T10:54:45.308761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb = XGBClassifier(random_state=1234)\nmodel7=xgb.fit(x_train, y_train)\nprint(\"train accuracy:\",model7.score(x_train, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:54:45.312411Z","iopub.execute_input":"2022-07-19T10:54:45.312783Z","iopub.status.idle":"2022-07-19T10:54:46.186811Z","shell.execute_reply.started":"2022-07-19T10:54:45.312748Z","shell.execute_reply":"2022-07-19T10:54:46.186007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/nlp-getting-started/test.csv')\ndef remove_URL(text):\n    url = re.compile(r'https?://\\S+')\n    return url.sub(r' httpsmark ', text)\n\n\ndef remove_html(text):\n    html = re.compile(r'<.*?>')\n    return html.sub(r'', text)\n\n\ndef remove_atsymbol(text):\n    name = re.compile(r'@\\S+')\n    return name.sub(r' atsymbol ', text)\n\n\ndef remove_hashtag(text):\n    hashtag = re.compile(r'#')\n    return hashtag.sub(r' hashtag ', text)\n\n\ndef remove_exclamation(text):\n    exclamation = re.compile(r'!')\n    return exclamation.sub(r' exclamation ', text)\n\n\ndef remove_question(text):\n    question = re.compile(r'?')\n    return question.sub(r' question ', text)\n\n\ndef remove_punc(text):\n    return text.translate(str.maketrans('','',string.punctuation))\n\n\ndef remove_number(text):\n    number = re.compile(r'\\d+')\n    return number.sub(r' number ', text)\n\n\ndef remove_emoji(string):\n    emoji_pattern = re.compile(\"[\"\n                               u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                               u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                               u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                               u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                               u\"\\U00002500-\\U00002BEF\"  # chinese char\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U000024C2-\\U0001F251\"\n                               u\"\\U0001f926-\\U0001f937\"\n                               u\"\\U00010000-\\U0010ffff\"\n                               u\"\\u2640-\\u2642\"\n                               u\"\\u2600-\\u2B55\"\n                               u\"\\u200d\"\n                               u\"\\u23cf\"\n                               u\"\\u23e9\"\n                               u\"\\u231a\"\n                               u\"\\ufe0f\"  # dingbats\n                               u\"\\u3030\"\n                               \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r' emoji ', string)\n\ntest['text'] = test['text'].str.lower()\ntest['text'] = test['text'].apply(lambda text: remove_URL(text))\ntest['text'] = test['text'].apply(lambda text: remove_html(text))\ntest['text'] = test['text'].apply(lambda text: remove_atsymbol(text))\ntest['text'] = test['text'].apply(lambda text: remove_hashtag(text))\ntest['text'] = test['text'].apply(lambda text: remove_exclamation(text))\ntest['text'] = test['text'].apply(lambda text: remove_punc(text))\ntest['text'] = test['text'].apply(lambda text: remove_number(text))\ntest['text'] = test['text'].apply(lambda text: remove_emoji(text))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:55:02.259239Z","iopub.execute_input":"2022-07-19T10:55:02.259638Z","iopub.status.idle":"2022-07-19T10:55:02.460982Z","shell.execute_reply.started":"2022-07-19T10:55:02.259606Z","shell.execute_reply":"2022-07-19T10:55:02.459776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(test.shape[0]):\n    words = word_tokenize(test['text'][i])\n    sentence = ' '.join([word for word in words if word not in stop_words])\n    test['text'][i] = sentence","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:55:09.612488Z","iopub.execute_input":"2022-07-19T10:55:09.613336Z","iopub.status.idle":"2022-07-19T10:55:11.787258Z","shell.execute_reply.started":"2022-07-19T10:55:09.613291Z","shell.execute_reply":"2022-07-19T10:55:11.786073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer = nltk.stem.PorterStemmer()\nfor i in range(test.shape[0]):\n    words = word_tokenize(test['text'][i])\n    sentence = ' '.join([stemmer.stem(word) for word in words if word not in stop_words])\n    test['text'][i] = sentence","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:55:14.023086Z","iopub.execute_input":"2022-07-19T10:55:14.023542Z","iopub.status.idle":"2022-07-19T10:55:17.663901Z","shell.execute_reply.started":"2022-07-19T10:55:14.023500Z","shell.execute_reply":"2022-07-19T10:55:17.662801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmatizer = WordNetLemmatizer()\nfor i in range(test.shape[0]):\n    words = word_tokenize(test['text'][i])\n    sentence = ' '.join([lemmatizer.lemmatize(word) for word in words if word not in stop_words])\n    test['text'][i] = sentence","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:55:19.520146Z","iopub.execute_input":"2022-07-19T10:55:19.520628Z","iopub.status.idle":"2022-07-19T10:55:21.815838Z","shell.execute_reply.started":"2022-07-19T10:55:19.520591Z","shell.execute_reply":"2022-07-19T10:55:21.814792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = vectorizer.transform(test['text'])\npred=model3.predict(x_test)\npred","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:55:26.352152Z","iopub.execute_input":"2022-07-19T10:55:26.352571Z","iopub.status.idle":"2022-07-19T10:55:26.431040Z","shell.execute_reply.started":"2022-07-19T10:55:26.352533Z","shell.execute_reply":"2022-07-19T10:55:26.430204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = { 'id':test.id, 'target':pred }\ndf = pd.DataFrame(df)\ndf.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:59:47.897971Z","iopub.execute_input":"2022-07-19T10:59:47.898442Z","iopub.status.idle":"2022-07-19T10:59:47.913701Z","shell.execute_reply.started":"2022-07-19T10:59:47.898403Z","shell.execute_reply":"2022-07-19T10:59:47.911979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}