{"cells":[{"metadata":{"_uuid":"9796a17d94f3c1879029f46ee8de6ab06626d614"},"cell_type":"markdown","source":"# Try the SVM NB method with quora text"},{"metadata":{"trusted":true,"_uuid":"c19432c641c99deee5c44e9f4c3b67b6fa3f916c"},"cell_type":"code","source":"import pandas as pd\nfrom fastai.imports import *\nfrom sklearn.feature_extraction.text import *\nfrom sklearn.metrics import *\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestRegressor, RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nimport re, string #import regular expression ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e554186cd492affc3056cae21261badb59b8ce7d"},"cell_type":"code","source":"#PATH = \"/media/cuberti/My Book/datascience/data/quora/\"\nPATH = '../input/' #file Input for running the Kaggle Kernel\ntrain = pd.read_csv(f\"{PATH}train.csv\")\ntest = pd.read_csv(f\"{PATH}test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42c594a331a1320277fa0716d9ef75b828a7804f"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a2e858faad92614eb12ecc92e7572e6de314011"},"cell_type":"code","source":"#Only run cell when training the data\n#train = train.sample(frac = 0.2, random_state = 42)\n#train,test,y_trn,y_tst = train_test_split(train.drop('target', axis = 1), train.target, test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e5ba161518ac943100fe90c7e7edeb08938767f"},"cell_type":"code","source":"#Otherwise run this cell for final submission\ny_trn = train.target\ntrain = train.drop('target', axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"193a22fef527699572de61412e95c51f869ce601"},"cell_type":"markdown","source":"### Create vecotrizer to create TDM and clean text"},{"metadata":{"trusted":true,"_uuid":"e40a3077a402d950a7d12957ad1050fb3598adcc"},"cell_type":"code","source":"re_tok = re.compile(f'([{string.punctuation}“”¨«»®´·º½¾¿¡§£₤‘’])') #look for weird punctuaition to remove\ndef tokenize(s): return re_tok.sub(r' \\1 ', s).lower().split() #split on spaces and substitue punctuation","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8f0db40affd2bee96174507257815a9bad0faa3"},"cell_type":"code","source":"%%time\nn = train.shape[0]\nvec = TfidfVectorizer(ngram_range=(1,2), tokenizer=tokenize, min_df=3, max_df=0.9, strip_accents='unicode', use_idf=1,\n               smooth_idf=1, sublinear_tf=1)\ntrn_tdm = vec.fit_transform(train['question_text'])\ntst_tdm = vec.transform(test['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcd1465bb439f3f815abf9c7c6f9f3f2dd36e610"},"cell_type":"code","source":"trn_tdm, tst_tdm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e53b2cd4d07f2d6a315f8c9909354c83e8ceaa2f"},"cell_type":"markdown","source":"## Try Naive Bayes features"},{"metadata":{"trusted":true,"_uuid":"a2c6d1e57fcc2b5c0b3995b6a63e8507b784c157"},"cell_type":"code","source":"def pr(x,y_i, y):\n    p = x[(y==y_i)].sum(0)\n    return (p+1) / ((y==y_i).sum()+1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7d7faa3d816c5aabc051aa12feed3db7dc631e8"},"cell_type":"code","source":"%%time\ny=y_trn.values\nr = np.log(pr(trn_tdm, 1,y) / pr(trn_tdm, 0,y))\nm=LogisticRegression(C=4, dual=True)\nm.fit(trn_tdm.multiply(r), y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acee30504ad320f0921c50674596a16d4aed17b4"},"cell_type":"code","source":"preds=m.predict_proba(tst_tdm.multiply(r))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a01f45f61b460cd7b919f270eea9e1d23061a916"},"cell_type":"code","source":"#skip this when submitting a final model\nn = 50\ni_list = []; f1_list=[];\nfor i in np.linspace(0,1, num = n+1):\n    i_list.append(i)\n    f1=f1_score(y_tst, preds[:,1]>i)\n    f1_list.append(f1)\ndf = pd.DataFrame({'i': i_list, 'f1':f1_list})\nimport matplotlib.pyplot as plt\nplt.scatter(x='i', y='f1', data = df)\nprint('Maximum Threshold \\n', df.iloc[df['f1'].idxmax()])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e9e83ee0f12204d3eff5d9617a0904cbd0e5628"},"cell_type":"code","source":"final_pred=(preds[:,1]>0.2).astype(int)\nmy_sub = pd.DataFrame({'qid':test.qid.values, 'prediction':final_pred})\nmy_sub.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"803b5498d91035df382f173b860c00a4f49ff6e6"},"cell_type":"code","source":"my_sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8361964719c7fdaaf6b80a9957df64719940d5c1"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}