{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T08:34:00.087118Z","iopub.execute_input":"2022-08-10T08:34:00.088485Z","iopub.status.idle":"2022-08-10T08:34:00.098067Z","shell.execute_reply.started":"2022-08-10T08:34:00.088397Z","shell.execute_reply":"2022-08-10T08:34:00.096528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\ntf.version.VERSION","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:00.230312Z","iopub.execute_input":"2022-08-10T08:34:00.231666Z","iopub.status.idle":"2022-08-10T08:34:00.238352Z","shell.execute_reply.started":"2022-08-10T08:34:00.231614Z","shell.execute_reply":"2022-08-10T08:34:00.237163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install demoji","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:00.338766Z","iopub.execute_input":"2022-08-10T08:34:00.339580Z","iopub.status.idle":"2022-08-10T08:34:11.409307Z","shell.execute_reply.started":"2022-08-10T08:34:00.339541Z","shell.execute_reply":"2022-08-10T08:34:11.407931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Manipulation libraries\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\n\n# NLP libraries\nimport string # Library for string operations\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem.snowball import SnowballStemmer\nimport re # Regex library\nimport demoji\nfrom wordcloud import WordCloud # Word Cloud library\n\n# ploting libraries\nimport matplotlib.pyplot as plt\n\n# ML/AI libraries\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn import svm","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:11.411986Z","iopub.execute_input":"2022-08-10T08:34:11.412432Z","iopub.status.idle":"2022-08-10T08:34:11.420990Z","shell.execute_reply.started":"2022-08-10T08:34:11.412387Z","shell.execute_reply":"2022-08-10T08:34:11.419888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#@title\ntrain_data = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ntest_data = pd.read_csv(\"../input/nlp-getting-started/test.csv\")\n\n# Basic Info\nprint(\"Columns: \", list(train_data.columns))\n\n# Features\nX_train = train_data[[\"id\", \"keyword\", \"location\", \"text\"]]\nX_test = test_data\n\n# Labels\ny_train = train_data[[\"id\",\"target\"]]\n\nprint(\"Training Data Size\", len(y_train))\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:11.422721Z","iopub.execute_input":"2022-08-10T08:34:11.423086Z","iopub.status.idle":"2022-08-10T08:34:11.487349Z","shell.execute_reply.started":"2022-08-10T08:34:11.423054Z","shell.execute_reply":"2022-08-10T08:34:11.486475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"## Analyzing labels","metadata":{}},{"cell_type":"code","source":"Real_len = train_data[train_data['target'] == 1].shape[0]\nNot_len = train_data[train_data['target'] == 0].shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:11.489607Z","iopub.execute_input":"2022-08-10T08:34:11.490229Z","iopub.status.idle":"2022-08-10T08:34:11.498686Z","shell.execute_reply.started":"2022-08-10T08:34:11.490195Z","shell.execute_reply":"2022-08-10T08:34:11.497171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bar plot of the 3 classes\nplt.rcParams['figure.figsize'] = (7, 5)\nplt.bar(10,Real_len,3, label=\"Real\", color='blue')\nplt.bar(15,Not_len,3, label=\"Not\", color='red')\nplt.legend()\nplt.ylabel('Number of examples')\nplt.title('Propertion of examples')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:11.502317Z","iopub.execute_input":"2022-08-10T08:34:11.503091Z","iopub.status.idle":"2022-08-10T08:34:11.713686Z","shell.execute_reply.started":"2022-08-10T08:34:11.503052Z","shell.execute_reply":"2022-08-10T08:34:11.712785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyzing Features","metadata":{}},{"cell_type":"markdown","source":"### Sentence Length Analysis","metadata":{}},{"cell_type":"code","source":"def length(string):    \n    return len(string)\ntrain_data['length'] = train_data['text'].apply(length)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:11.715071Z","iopub.execute_input":"2022-08-10T08:34:11.715638Z","iopub.status.idle":"2022-08-10T08:34:11.725586Z","shell.execute_reply.started":"2022-08-10T08:34:11.715603Z","shell.execute_reply":"2022-08-10T08:34:11.724510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = (18.0, 6.0)\nbins = 150\nplt.hist(train_data[train_data['target'] == 0]['length'], alpha = 0.6, bins=bins, label='Not')\nplt.hist(train_data[train_data['target'] == 1]['length'], alpha = 0.8, bins=bins, label='Real')\nplt.xlabel('length')\nplt.ylabel('numbers')\nplt.legend(loc='upper right')\nplt.xlim(0,150)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:11.727697Z","iopub.execute_input":"2022-08-10T08:34:11.728759Z","iopub.status.idle":"2022-08-10T08:34:12.718691Z","shell.execute_reply.started":"2022-08-10T08:34:11.728721Z","shell.execute_reply":"2022-08-10T08:34:12.717503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len=train_data[train_data['target']==1]['text'].str.len()\nax1.hist(tweet_len,color='blue')\nax1.set_title('disaster tweets')\ntweet_len=train_data[train_data['target']==0]['text'].str.len()\nax2.hist(tweet_len,color='red')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:12.720196Z","iopub.execute_input":"2022-08-10T08:34:12.721739Z","iopub.status.idle":"2022-08-10T08:34:13.042924Z","shell.execute_reply.started":"2022-08-10T08:34:12.721701Z","shell.execute_reply":"2022-08-10T08:34:13.041571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"**Data cleaning is the process of preparing data for analysis by removing or modifying data that is incorrect, incomplete, irrelevant, duplicated, or improperly formatted.**\n\n1. Remove Url\n2. Handle Tags\n3. Handle emoji's\n4. Remove HTML Tags\n5. Remove stopwords\n6. Removing Useless Characters","metadata":{}},{"cell_type":"code","source":"# Step 1. Remove Url\n#https://stackoverflow.com/questions/11331982/how-to-remove-any-url-within-a-string-in-python/11332580\ndef Remove_Url(string):\n    return re.sub(r'(https|http)?:\\/\\/(\\w|\\.|\\/|\\?|\\=|\\&|\\%|\\-)*\\b', '', string)\n\n# Step 2. Handle Tags\ndef Handle_Tags(string):\n    pattern = re.compile(r'[@|#][^\\s]+')\n    matches = pattern.findall(string)\n    tags = [match[1:] for match in matches]\n    # Removing tags from main string\n    string = re.sub(pattern, '', string)\n    # More weightage to tag by adding them 3 times\n    return string + ' ' + ' '.join(tags) + ' '+ ' '.join(tags) + ' ' + ' '.join(tags)\n\n# Step 3. Handle emoji's\n#http://unicode.org/Public/emoji/12.0/emoji-test.txt\ndemoji.download_codes()\ndef Handle_emoji(string):\n    return demoji.replace_with_desc(string)\n\n# Step 4. Remove HTML Tags\ndef Remove_html(string):\n    return re.sub(r'<.*?>|&([a-z0-9]+|#[0-9]{1,6}|#x[0-9a-f]{1,6});', '', str(string))\n\n# Step 5. Remove Stopwords and Stemming\nnltk.download('punkt')\nnltk.download('stopwords')\nstemmer  = SnowballStemmer('english')\nstopword = stopwords.words('english')\ndef Remove_StopAndStem(string):\n    string_list = string.split()\n    return ' '.join([stemmer.stem(i) for i in string_list if i not in stopword])\n\n# Step 6. Removing Useless Characters\ndef Remove_UC(string):\n    thestring = re.sub(r'[^a-zA-Z\\s]','', string)\n    # remove word of length less than 2\n    thestring = re.sub(r'\\b\\w{1,2}\\b', '', thestring)\n    #https://www.geeksforgeeks.org/python-remove-unwanted-spaces-from-string/\n    return re.sub(' +', ' ', thestring) \n\n# Step7. Merging Other Details\ndef merging_details(data):\n        #df = pd.DataFrame(columns=['id', 'Cleaned_data'])\n        df_list = []\n        \n        #https://www.geeksforgeeks.org/how-to-iterate-over-rows-in-pandas-dataframe/\n        for row in data.itertuples():\n            df_dict = {}\n            # Processing Keyword and location\n            keyword = re.sub(r'[^a-zA-Z\\s]','', str(row[2]))\n            location = re.sub(r'[^a-zA-Z\\s]','', str(row[3]))\n            keyword = re.sub(r'\\b\\w{1,2}\\b', '', keyword)\n            location = re.sub(r'\\b\\w{1,2}\\b', '', location)\n            # Already processed data\n            text = str(row[4])\n\n            if keyword == 'nan':\n                if location == 'nan':    \n                    prs_data = text\n                else:\n                    prs_data = location + ' ' + text\n            else:\n                if location == 'nan':    \n                    prs_data = keyword + ' ' + text\n                else:\n                    prs_data = keyword + ' ' + location + ' ' + text                \n            \n            prs_data = re.sub(' +', ' ', prs_data) \n            \n            df_dict['Cleaned_data'] = prs_data\n            \n            df_list.append(df_dict)\n                 \n        return pd.DataFrame(df_list)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:13.045851Z","iopub.execute_input":"2022-08-10T08:34:13.046744Z","iopub.status.idle":"2022-08-10T08:34:13.064841Z","shell.execute_reply.started":"2022-08-10T08:34:13.046690Z","shell.execute_reply":"2022-08-10T08:34:13.063738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Pre-Processing Data","metadata":{}},{"cell_type":"code","source":"# Step 1. Remove Url\nX_train['text'] = X_train['text'].apply(Remove_Url)\nX_test['text'] = X_test['text'].apply(Remove_Url)\n\n# Step 2. Handle Tags\nX_train['text'] = X_train['text'].apply(Handle_Tags)\nX_test['text'] = X_test['text'].apply(Handle_Tags)\n\n# Step 3. Handle emoji's\nX_train['text'] = X_train['text'].apply(Handle_emoji)\nX_test['text'] = X_test['text'].apply(Handle_emoji)\n\n# Step 4. Remove HTML Tags\nX_train['text'] = X_train['text'].apply(Remove_html)\nX_test['text'] = X_test['text'].apply(Remove_html)\n\n# Step 5. Remove Stopwords and Stemming\nX_train['text'] = X_train['text'].apply(Remove_StopAndStem)\nX_test['text'] = X_test['text'].apply(Remove_StopAndStem)\n\n# Step 6. Removing Useless Characters\nX_train['text'] = X_train['text'].apply(Remove_UC)\nX_test['text'] = X_test['text'].apply(Remove_UC)\n\n# Step7. Merging Other Details\nX_train = merging_details(X_train)\nX_test = merging_details(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:13.068120Z","iopub.execute_input":"2022-08-10T08:34:13.068533Z","iopub.status.idle":"2022-08-10T08:34:23.695301Z","shell.execute_reply.started":"2022-08-10T08:34:13.068497Z","shell.execute_reply":"2022-08-10T08:34:23.694106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## WORDCLOUD","metadata":{}},{"cell_type":"code","source":"%%time\ndict_of_words = {}\nfor row in  X_train.itertuples():\n    for i in row[1].split():\n        try:\n            dict_of_words[i] += 1\n        except:\n            dict_of_words[i] = 1\n\n#Initializing  WordCloud\nwordcloud = WordCloud(background_color = 'black', width=1000, height=500).generate_from_frequencies(dict_of_words)\nfig = plt.figure(figsize=(10,5))\nplt.imshow(wordcloud)\nplt.tight_layout(pad=1)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:23.696967Z","iopub.execute_input":"2022-08-10T08:34:23.697314Z","iopub.status.idle":"2022-08-10T08:34:25.209780Z","shell.execute_reply.started":"2022-08-10T08:34:23.697284Z","shell.execute_reply":"2022-08-10T08:34:25.208784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Extraction","metadata":{}},{"cell_type":"code","source":"%%time\n#smooth_idf=True by default so smoothing is done by defult.\n#norm is l2 by default.\n#subliner is used False by default.\nvectorizer = TfidfVectorizer(min_df = 0.0005, \n                             max_features = 100000, \n                             tokenizer = lambda x: x.split(),\n                             ngram_range = (1,4))\n\n\nX_train = vectorizer.fit_transform(X_train['Cleaned_data'])\nX_test = vectorizer.transform(X_test['Cleaned_data'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:25.211169Z","iopub.execute_input":"2022-08-10T08:34:25.211759Z","iopub.status.idle":"2022-08-10T08:34:26.349810Z","shell.execute_reply.started":"2022-08-10T08:34:25.211723Z","shell.execute_reply":"2022-08-10T08:34:26.348517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Classification using SVM","metadata":{}},{"cell_type":"code","source":"%%time\nModel = svm.SVC(kernel='linear')\nModel.fit(X_train, y_train['target'])\ny_pred = Model.predict(X_test)\npred = y_pred.round().astype('int32')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:26.351430Z","iopub.execute_input":"2022-08-10T08:34:26.352311Z","iopub.status.idle":"2022-08-10T08:34:32.867290Z","shell.execute_reply.started":"2022-08-10T08:34:26.352260Z","shell.execute_reply":"2022-08-10T08:34:32.865718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating submission.csv to publish results\nsubmission = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\nsubmission['target'] = pred\nsubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T08:34:32.870743Z","iopub.execute_input":"2022-08-10T08:34:32.872079Z","iopub.status.idle":"2022-08-10T08:34:32.887469Z","shell.execute_reply.started":"2022-08-10T08:34:32.872026Z","shell.execute_reply":"2022-08-10T08:34:32.886350Z"},"trusted":true},"execution_count":null,"outputs":[]}]}