{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score, classification_report, accuracy_score, log_loss, roc_auc_score\nimport scipy\nimport xgboost as xgb\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\nsubm = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f5725da90da0a3fd90b5444bf17ac04f2d03ffe5"},"cell_type":"markdown","source":"##  Looking into Data"},{"metadata":{"trusted":true,"_uuid":"8bc9d3af9af8b2ce4fa4c7eda386d60609b457b1"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac184e0145ce389ab870c129166fccd397c16278"},"cell_type":"code","source":"lens = train.question_text.str.len()\nlens.mean(), lens.std(), lens.max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6abae45b973d5539c06404de669cecb37766960e"},"cell_type":"code","source":"lens.hist()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"22e3e3c17beef4af81552ed15888541fbd0cfb10"},"cell_type":"markdown","source":"Create a 'none' label so we can see how many questions have no labels. We can then summarize the dataset."},{"metadata":{"trusted":true,"_uuid":"085297cc39c8007ab3b420b7902d3e284bdc9106"},"cell_type":"code","source":"label_cols = ['target']\ntrain['none'] = 1-train[label_cols].max(axis=1)\ntrain.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a46194f7447350cf26ccf1009af6170543e765c6"},"cell_type":"code","source":"len(train),len(test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0d5cf0609b3c30e2373a0efe2a1230549fa0ff77"},"cell_type":"markdown","source":"So it looks like there is no empty questions. "},{"metadata":{"trusted":true,"_uuid":"4f5c1625ed83e5415f4454b4d23e92265d7eb59c"},"cell_type":"markdown","source":"## Text Cleaning\nFollowing the suggestion [here](https://towardsdatascience.com/finding-similar-quora-questions-with-bow-tfidf-and-random-forest-c54ad88d1370), text cleaning follows this approach\n\n1. Not to remove stop words, because words like “what”, “which” and “how” may have strong signals.\n2. Not to stem words.\n3. Remove punctuation.\n4. Correct typos.\n5. Change abbreviations to its original terms.\n6. Remove comma between numbers.\n7. Change special chars to words."},{"metadata":{"trusted":true,"_uuid":"b10b73b3082b92c21c4a650c57437bffc975b9c1"},"cell_type":"code","source":"SPECIAL_TOKENS = {\n    'quoted': 'quoted_item',\n    'non-ascii': 'non_ascii_word',\n    'undefined': 'something'\n}\n\ndef clean(text, stem_words=True):\n    import re\n    from string import punctuation\n    from nltk.stem import SnowballStemmer\n    from nltk.corpus import stopwords\n    \n    def pad_str(s):\n        return ' '+s+' '\n    \n    if pd.isnull(text):\n        return ''\n\n    #    stops = set(stopwords.words(\"english\"))\n    # Clean the text, with the option to stem words.\n    \n    # Empty question\n    \n    if type(text) != str or text=='':\n        return ''\n\n    # Clean the text\n    text = re.sub(\"\\'s\", \" \", text) \n    text = re.sub(\" whats \", \" what is \", text, flags=re.IGNORECASE)\n    text = re.sub(\"\\'ve\", \" have \", text)\n    text = re.sub(\"can't\", \"can not\", text)\n    text = re.sub(\"n't\", \" not \", text)\n    text = re.sub(\"i'm\", \"i am\", text, flags=re.IGNORECASE)\n    text = re.sub(\"\\'re\", \" are \", text)\n    text = re.sub(\"\\'d\", \" would \", text)\n    text = re.sub(\"\\'ll\", \" will \", text)\n    text = re.sub(\"e\\.g\\.\", \" eg \", text, flags=re.IGNORECASE)\n    text = re.sub(\"b\\.g\\.\", \" bg \", text, flags=re.IGNORECASE)\n    text = re.sub(\"(\\d+)(kK)\", \" \\g<1>000 \", text)\n    text = re.sub(\"e-mail\", \" email \", text, flags=re.IGNORECASE)\n    text = re.sub(\"(the[\\s]+|The[\\s]+)?U\\.S\\.A\\.\", \" America \", text, flags=re.IGNORECASE)\n    text = re.sub(\"(the[\\s]+|The[\\s]+)?United State(s)?\", \" America \", text, flags=re.IGNORECASE)\n    text = re.sub(\"\\(s\\)\", \" \", text, flags=re.IGNORECASE)\n    text = re.sub(\"[c-fC-F]\\:\\/\", \" disk \", text)\n    \n    # remove comma between numbers, i.e. 15,000 -> 15000\n    \n    text = re.sub('(?<=[0-9])\\,(?=[0-9])', \"\", text)\n    \n    \n    # add padding to punctuations and special chars, we still need them later\n    \n    text = re.sub('\\$', \" dollar \", text)\n    text = re.sub('\\%', \" percent \", text)\n    text = re.sub('\\&', \" and \", text)\n\n        \n    text = re.sub('[^\\x00-\\x7F]+', pad_str(SPECIAL_TOKENS['non-ascii']), text) # replace non-ascii word with special word\n    \n    # indian dollar\n    \n    text = re.sub(\"(?<=[0-9])rs \", \" rs \", text, flags=re.IGNORECASE)\n    text = re.sub(\" rs(?=[0-9])\", \" rs \", text, flags=re.IGNORECASE)\n    \n    # clean text rules get from : https://www.kaggle.com/currie32/the-importance-of-cleaning-text\n    text = re.sub(r\" (the[\\s]+|The[\\s]+)?US(A)? \", \" America \", text)\n    text = re.sub(r\" UK \", \" England \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" india \", \" India \", text)\n    text = re.sub(r\" switzerland \", \" Switzerland \", text)\n    text = re.sub(r\" china \", \" China \", text)\n    text = re.sub(r\" chinese \", \" Chinese \", text) \n    text = re.sub(r\" imrovement \", \" improvement \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" intially \", \" initially \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" quora \", \" Quora \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" dms \", \" direct messages \", text, flags=re.IGNORECASE)  \n    text = re.sub(r\" demonitization \", \" demonetization \", text, flags=re.IGNORECASE) \n    text = re.sub(r\" actived \", \" active \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" kms \", \" kilometers \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" cs \", \" computer science \", text, flags=re.IGNORECASE) \n    text = re.sub(r\" upvote\", \" up vote\", text, flags=re.IGNORECASE)\n    text = re.sub(r\" iPhone \", \" phone \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" \\0rs \", \" rs \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" calender \", \" calendar \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" ios \", \" operating system \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" gps \", \" GPS \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" gst \", \" GST \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" programing \", \" programming \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" bestfriend \", \" best friend \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" dna \", \" DNA \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" III \", \" 3 \", text)\n    text = re.sub(r\" banglore \", \" Banglore \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" J K \", \" JK \", text, flags=re.IGNORECASE)\n    text = re.sub(r\" J\\.K\\. \", \" JK \", text, flags=re.IGNORECASE)\n    \n    # replace the float numbers with a random number, it will be parsed as number afterward, and also been replaced with word \"number\"\n    \n    text = re.sub('[0-9]+\\.[0-9]+', \" 87 \", text)\n    \n    # Remove punctuation from text\n    text = ''.join([c for c in text if c not in punctuation]).lower()\n       # Return a list of words\n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d63e258aed27fda465c0cf2046cc2893c494935"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(clean)\ntest['question_text'] = test['question_text'].apply(clean)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ce833bced99940f992f3a2f99c0ebc9581ce8ae"},"cell_type":"markdown","source":"## Building the Model\n\n### Character level TF-IDF + XGBOOST"},{"metadata":{"trusted":true,"_uuid":"f2641a2aea922d51aa56e21f2b49467bcfb32a42"},"cell_type":"code","source":"# prepare the input for model\n# Referencd https://github.com/susanli2016/NLP-with-Python/blob/master/BOW_TFIDF_Xgboost_update.ipynb\n\ntfidf_vect_ngram_chars = TfidfVectorizer(analyzer='char', token_pattern=r'\\w{1,}', ngram_range=(2,3), max_features=5000)\ntrain_transform = tfidf_vect_ngram_chars.fit_transform(train['question_text'].values)\ntest_transform = tfidf_vect_ngram_chars.transform(test['question_text'].values)\n\nX = train_transform\ny = train['target'].values\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d7e6734325fc9218ed9b8637a9aa72a60371a40"},"cell_type":"code","source":"# split the train set to train and valid\n\nX_train,X_valid,y_train,y_valid = train_test_split(X,y, test_size = 0.33, random_state = 42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b81373d52eefa9b44824866849fc3bd97e34037"},"cell_type":"code","source":"xgb_model = xgb.XGBClassifier(max_depth=50, n_estimators=80, learning_rate=0.1, colsample_bytree=.7, gamma=0, reg_alpha=4, objective='binary:logistic', eta=0.3, silent=1, subsample=0.8).fit(X_train, y_train)\nxgb_prediction = xgb_model.predict(X_valid)\nprint('n-gram level tf-idf training score:', f1_score(y_train, xgb_model.predict(X_train), average='macro'))\nprint('n-gram level tf-idf validation score:', f1_score(y_valid, xgb_model.predict(X_valid), average='macro'))\nprint(classification_report(y_valid, xgb_prediction))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6aa2893e35974e6e584aa7b0fa1acb216afe84e1"},"cell_type":"code","source":"#gbm = xgb.XGBClassifier(max_depth=3, n_estimators=300, learning_rate=0.05).fit(X, y)\n#predictions = gbm.predict(test_transform)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54b32e0fb5e899406661952143a5493871133ddd"},"cell_type":"code","source":"# do the submission\n\nsubmission = pd.DataFrame({ 'qid': test['qid'],\n                            'prediction': xgb_model.predict(test_transform) })\nsubmission.to_csv(\"submission.csv\", index=False)\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}