{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T15:44:21.721419Z","iopub.execute_input":"2022-07-28T15:44:21.722490Z","iopub.status.idle":"2022-07-28T15:44:21.730934Z","shell.execute_reply.started":"2022-07-28T15:44:21.722443Z","shell.execute_reply":"2022-07-28T15:44:21.729767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest=pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsample=pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:54:32.220279Z","iopub.execute_input":"2022-07-28T17:54:32.221465Z","iopub.status.idle":"2022-07-28T17:54:32.257927Z","shell.execute_reply.started":"2022-07-28T17:54:32.221417Z","shell.execute_reply":"2022-07-28T17:54:32.256944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom nltk.tokenize import word_tokenize,sent_tokenize\nfrom nltk.stem import PorterStemmer,WordNetLemmatizer\nfrom nltk.corpus import stopwords\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.metrics import accuracy_score,classification_report","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:32:39.658807Z","iopub.execute_input":"2022-07-28T17:32:39.659155Z","iopub.status.idle":"2022-07-28T17:32:39.665199Z","shell.execute_reply.started":"2022-07-28T17:32:39.659126Z","shell.execute_reply":"2022-07-28T17:32:39.664004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic approach using below mentioned techniques\n### **NOTE** :In all machine learning approaches we will use only Naive Bayes Classifier\n## Text preprocessing stage 1\n1. Sanity cleaning(using regex)\n2. Stopwords\n3. Stemming\n4. Lemmatization\n\n## Text preprocessing stage 2 (Text modelling)\n1. Bag of Words\n2. TF-IDF","metadata":{}},{"cell_type":"code","source":"def text_preprocess(corpus):\n    result_corpus=[]\n    corpus=corpus.to_list()\n    for sentence in corpus:\n        temp_sen=re.sub('[^A-Za-z]',' ',sentence)\n        result_corpus.append(temp_sen.lower().strip())\n    return result_corpus\ndef preprocess_basic(df:pd.DataFrame,technique:str,test=None):\n    corpus=df.text\n    if test!=True:\n        df.drop(columns=['id','target'],inplace=True)\n    else:\n        df.drop(columns=['id'],inplace=True)\n    df.keyword.fillna('others',inplace=True)\n    common_loc=list(train.location.value_counts().head(10).to_dict().keys())\n    df.location=df.location.apply(lambda x: x if x in common_loc else 'others' )\n    dftmp=pd.get_dummies(df,columns=['keyword','location'])\n    clean_corpus=text_preprocess(corpus.copy())\n    if technique=='lemmatization':\n        clean_corpus=lemmatize(clean_corpus)\n    elif technique=='stemming':\n        clean_corpus=stemmed(clean_corpus)\n    return dftmp,clean_corpus\ndef stemmed(corpus:list):\n    result=[]\n    stem=PorterStemmer()\n    for word in corpus:\n        temp=[stem.stem(w) for w in word.split(' ') if w not in stopwords.words('english')]\n        result.append(' '.join(temp).strip().lower())\n    return result\ndef lemmatize(corpus:list):\n    result=[]\n    stem=WordNetLemmatizer()\n    for word in corpus:\n        temp=[stem.lemmatize(w) for w in word.split(' ') if w not in stopwords.words('english')]\n        result.append(' '.join(temp).strip().lower())\n    return result\ndef train_and_validate(X,y):\n    X_train,X_val,y_train,y_val=train_test_split(X,y,test_size=0.2)\n    model = MultinomialNB()\n    model.fit(X_train,y_train)\n    y_preds=model.predict(X_val)\n    print(\"Accuracy is :\",accuracy_score(y_val,y_preds))\n    print(classification_report(y_val,y_preds))\ndef fully_train(X,y):\n    model = MultinomialNB()\n    model.fit(X,y)\n    print('Training done on whole dataset')\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:59:10.020433Z","iopub.execute_input":"2022-07-28T17:59:10.020800Z","iopub.status.idle":"2022-07-28T17:59:10.036902Z","shell.execute_reply.started":"2022-07-28T17:59:10.020770Z","shell.execute_reply":"2022-07-28T17:59:10.035879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bag of Words + Stemming","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.model_selection import train_test_split\ntraindf,traincorpus=preprocess_basic(train.copy(),'stemming')\ncv=CountVectorizer(ngram_range=(2,3),max_features=1000)\nX=cv.fit_transform(traincorpus).toarray()\nX=np.concatenate([X,traindf.drop(columns=['text']).to_numpy()],axis=1)\ny=train.target\ntrain_and_validate(X,y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:55:45.305152Z","iopub.execute_input":"2022-07-28T17:55:45.305760Z","iopub.status.idle":"2022-07-28T17:56:08.038157Z","shell.execute_reply.started":"2022-07-28T17:55:45.305716Z","shell.execute_reply":"2022-07-28T17:56:08.037155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bag of Words + Lemmatization","metadata":{}},{"cell_type":"code","source":"traindf,traincorpus=preprocess_basic(train.copy(),'lemmatize')\ncv=CountVectorizer(ngram_range=(2,3),max_features=1000)\nX=cv.fit_transform(traincorpus).toarray()\nX=np.concatenate([X,traindf.drop(columns=['text']).to_numpy()],axis=1)\ny=train.target\ntrain_and_validate(X,y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:56:33.511422Z","iopub.execute_input":"2022-07-28T17:56:33.511803Z","iopub.status.idle":"2022-07-28T17:56:34.481158Z","shell.execute_reply.started":"2022-07-28T17:56:33.511770Z","shell.execute_reply":"2022-07-28T17:56:34.480157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### At this point we already have two approaches for solution , according to results BOW+Lemmatization performed better we can train whole dataset on that and get final submission file","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:29:40.391357Z","iopub.execute_input":"2022-07-28T17:29:40.391697Z","iopub.status.idle":"2022-07-28T17:29:40.398605Z","shell.execute_reply.started":"2022-07-28T17:29:40.391669Z","shell.execute_reply":"2022-07-28T17:29:40.397217Z"}}},{"cell_type":"code","source":"traindf,traincorpus=preprocess_basic(train.copy(),'lemmatize')\ncv=CountVectorizer(ngram_range=(2,3),max_features=1000)\nX=cv.fit_transform(traincorpus).toarray()\nX=np.concatenate([X,traindf.drop(columns=['text']).to_numpy()],axis=1)\ny=train.target\nmymodel=fully_train(X,y)\ntestdf,testcorpus=preprocess_basic(test.copy(),'lemmatize',True)\ncv=CountVectorizer(ngram_range=(2,3),max_features=1000)\nX_test=cv.fit_transform(testcorpus).toarray()\nX_test=np.concatenate([X_test,testdf.drop(columns=['text']).to_numpy()],axis=1)\nresults=mymodel.predict(X_test)\nsubmission=pd.DataFrame()\nsubmission['id']=test.id\nsubmission['target']=results\nsubmission.to_csv('bow_stemming.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T17:59:16.001852Z","iopub.execute_input":"2022-07-28T17:59:16.002584Z","iopub.status.idle":"2022-07-28T17:59:17.678208Z","shell.execute_reply.started":"2022-07-28T17:59:16.002545Z","shell.execute_reply":"2022-07-28T17:59:17.677166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### TF-IDF + Lemmatization","metadata":{}},{"cell_type":"code","source":"#TFIDF\nfrom sklearn.feature_extraction.text import TfidfVectorizer\ncv=TfidfVectorizer(ngram_range=(1,1),max_features=10000)\nX=cv.fit_transform(lemmatized)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:06:55.395938Z","iopub.execute_input":"2022-07-27T16:06:55.396346Z","iopub.status.idle":"2022-07-27T16:06:55.595860Z","shell.execute_reply.started":"2022-07-27T16:06:55.396314Z","shell.execute_reply":"2022-07-27T16:06:55.594765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:14:53.908909Z","iopub.execute_input":"2022-07-27T16:14:53.909955Z","iopub.status.idle":"2022-07-27T16:14:53.917204Z","shell.execute_reply.started":"2022-07-27T16:14:53.909907Z","shell.execute_reply":"2022-07-27T16:14:53.916032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=train.target","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:09:41.824320Z","iopub.execute_input":"2022-07-27T16:09:41.824861Z","iopub.status.idle":"2022-07-27T16:09:41.834208Z","shell.execute_reply.started":"2022-07-27T16:09:41.824815Z","shell.execute_reply":"2022-07-27T16:09:41.832205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:08:03.833384Z","iopub.execute_input":"2022-07-27T16:08:03.833934Z","iopub.status.idle":"2022-07-27T16:08:03.848167Z","shell.execute_reply.started":"2022-07-27T16:08:03.833887Z","shell.execute_reply":"2022-07-27T16:08:03.846973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:09:01.056360Z","iopub.execute_input":"2022-07-27T16:09:01.056889Z","iopub.status.idle":"2022-07-27T16:09:01.062037Z","shell.execute_reply.started":"2022-07-27T16:09:01.056829Z","shell.execute_reply":"2022-07-27T16:09:01.060689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=GaussianNB()\nmodel.fit(X.toarray(),y)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:10:17.194929Z","iopub.execute_input":"2022-07-27T16:10:17.195447Z","iopub.status.idle":"2022-07-27T16:10:18.788810Z","shell.execute_reply.started":"2022-07-27T16:10:17.195404Z","shell.execute_reply":"2022-07-27T16:10:18.787790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:11:03.728292Z","iopub.execute_input":"2022-07-27T16:11:03.728800Z","iopub.status.idle":"2022-07-27T16:11:03.746180Z","shell.execute_reply.started":"2022-07-27T16:11:03.728754Z","shell.execute_reply":"2022-07-27T16:11:03.744989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testcorpus=test.text.to_list()\nstem=WordNetLemmatizer()\nlemmatizedt=[]\ncorpus_cleanedt=[]\nfor doc in testcorpus:\n    temp=re.sub('[^A-Za-z]',' ',doc)\n    temp=''.join(temp)\n    corpus_cleanedt.append(temp)\nfor doc in corpus_cleanedt:\n    words=[stem.lemmatize(word).lower().strip() for word in doc.split(' ') if word not in stopwords.words('english')]\n    words=' '.join(words)\n    lemmatizedt.append(words.strip())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:13:31.140240Z","iopub.execute_input":"2022-07-27T16:13:31.140609Z","iopub.status.idle":"2022-07-27T16:13:41.741310Z","shell.execute_reply.started":"2022-07-27T16:13:31.140578Z","shell.execute_reply":"2022-07-27T16:13:41.740268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TFIDF\nfrom sklearn.feature_extraction.text import TfidfVectorizer\ncv=TfidfVectorizer(ngram_range=(1,1),max_features=10000)\nXtest=cv.fit_transform(lemmatizedt)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:14:18.606619Z","iopub.execute_input":"2022-07-27T16:14:18.607659Z","iopub.status.idle":"2022-07-27T16:14:19.059597Z","shell.execute_reply.started":"2022-07-27T16:14:18.607619Z","shell.execute_reply":"2022-07-27T16:14:19.058515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=model.predict(Xtest.toarray())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:17:07.083330Z","iopub.execute_input":"2022-07-27T16:17:07.083886Z","iopub.status.idle":"2022-07-27T16:17:07.819328Z","shell.execute_reply.started":"2022-07-27T16:17:07.083841Z","shell.execute_reply":"2022-07-27T16:17:07.818249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.DataFrame()\nsubmission['id']=test.id\nsubmission['target']=preds","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:18:13.589530Z","iopub.execute_input":"2022-07-27T16:18:13.589944Z","iopub.status.idle":"2022-07-27T16:18:13.598703Z","shell.execute_reply.started":"2022-07-27T16:18:13.589911Z","shell.execute_reply":"2022-07-27T16:18:13.597225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:27:03.858866Z","iopub.execute_input":"2022-07-27T16:27:03.859251Z","iopub.status.idle":"2022-07-27T16:27:03.873766Z","shell.execute_reply.started":"2022-07-27T16:27:03.859219Z","shell.execute_reply":"2022-07-27T16:27:03.872384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}