{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nsub= pd.read_csv('../input/sample_submission.csv')\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)\nprint(\"sub : \", sub.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"324254f314349756b9649979563ffae52f8a9411"},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3206fa03dadad6af5ea1bbf2c2c7b55c6a6b9dea"},"cell_type":"code","source":"## split to train and val\ntrain, val = train_test_split(train, test_size=0.08, random_state=2018)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb69590c40dc9740d768ef29708f49b722af86d2"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f4f70e4793961cc045b49d62674b24a750b4fa60"},"cell_type":"code","source":"train['question_text'][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4739659bbffc2260d459415014a554422d4ff65f"},"cell_type":"code","source":"lens = train.question_text.str.len()\nlens.mean(), lens.std(), lens.max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de766505ba807581b66b8a86c8ccf797e95a61e8"},"cell_type":"code","source":"from matplotlib import pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aecd99c99bf99872d8f4beefe6b261eff34f375c"},"cell_type":"code","source":"lens.hist();\nplt.title('Counts for different length of questions')\nplt.ylabel('Count')\nplt.xlabel('Length of questions')\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c3c298945ba8ef9d31b8953adff27ca9257460fb"},"cell_type":"markdown","source":"The tail is quite long, let's take a look how does it looks like."},{"metadata":{"trusted":true,"_uuid":"5460ae520abc079c50f0817718fc164e82be960a"},"cell_type":"code","source":"train.loc[lens.argmax()]['question_text']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"05872d76ff2be79340f23690d271af9beea72481"},"cell_type":"markdown","source":"So it is a question about math with Latex. :) I am gonna keep it as what it is.... but a bit processing for these Latex syntax may become a good feature"},{"metadata":{"trusted":true,"_uuid":"b031b779798f5031a61d99ea18cf1dd13883ce61"},"cell_type":"code","source":"train.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de401cc693abdc7d573e9669da107ae8525868b9"},"cell_type":"code","source":"import re, string\nre_tok = re.compile(f'([{string.punctuation}“”¨«»®´·º½¾¿¡§£₤‘’\u0010])')\ndef tokenize(s): return re_tok.sub(r' \\1 ', s).split()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a76d48ea596c29fd236f0464115c483a369e7a36"},"cell_type":"markdown","source":"Lets do a TF-IDF for linear features"},{"metadata":{"trusted":true,"_uuid":"73caf718767b5ad49c98ce2612728e942e810521"},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bfb87e463901c1e5188dee858d9860fa6c18df44"},"cell_type":"code","source":"n = train.shape[0]\nvec = TfidfVectorizer(ngram_range=(1,2), tokenizer=tokenize,\n               min_df=3, max_df=0.9, strip_accents='unicode', use_idf=1,\n               smooth_idf=1, sublinear_tf=1 )\ntrn_term_doc = vec.fit_transform(train['question_text'])\nval_term_doc = vec.transform(val['question_text'])\ntest_term_doc = vec.transform(test['question_text'])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7332a3c84246eabec616d057c80895fb89dd9cd7"},"cell_type":"code","source":"vec_count = CountVectorizer(ngram_range=(1,2), tokenizer=tokenize,\n              strip_accents='unicode' )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8de6a3df4d5f6db55df6d1b2a5a37a7da8d0d15e"},"cell_type":"code","source":"trn_term_doc_count = vec_count.fit_transform(train['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5dcae9480bce61760d4ca8a957c70111d59b1bf4"},"cell_type":"code","source":"word_dict = vec_count.vocabulary_\nword_df = pd.DataFrame.from_dict(word_dict, orient='index')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"fa92f54369f4df3547208064f0ebffe1fe9b4318"},"cell_type":"code","source":"word_df = word_df.rename({0:'Count'}, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd776987ef21e2c93257a1aa4010935ee86fcf6c"},"cell_type":"code","source":"word_df.sort_values('Count').head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c66c3e1f2e21a99f9cfb11e4a76f84db7a63eb9e"},"cell_type":"code","source":"trn_term_doc, val_term_doc, test_term_doc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbc96cef2c67a4303d9c7a09088ca212754fd62a"},"cell_type":"code","source":"def pr(y_i, y):\n    p = x[y==y_i].sum(0)\n    return (p+1) / ((y==y_i).sum()+1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1284044eb39a2747c5fd10fdefecdcf28308bcb"},"cell_type":"code","source":"x = trn_term_doc\nval_x = val_term_doc\ntest_x = test_term_doc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb658eaf054df78c6e47c67f1ce611c497c95453"},"cell_type":"code","source":"def get_mdl(y):\n    y = y.values\n    r = np.log(pr(1,y) / pr(0,y))\n    m = LogisticRegression(C=4, dual=True)\n    x_nb = x.multiply(r)\n    return m.fit(x_nb, y), r","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f772c4d476d1d710346cd4960129d359f86925c1"},"cell_type":"markdown","source":"# VALIDATION SCORE"},{"metadata":{"trusted":true,"_uuid":"c32dabb2c1245cb9c2cade4f75975b76f0075b99"},"cell_type":"code","source":"preds = np.zeros((len(val), 1))\nm,r = get_mdl(train['target'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5636dba93dcc33c27beeec0f01579245bc61ce4e"},"cell_type":"code","source":"preds[:,0] = m.predict_proba(val_x.multiply(r))[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"accbe2c80c3d82e7e827bcd51eff3b2d5e0aad9c"},"cell_type":"code","source":"preds.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a44908f4c5e2901db08ea35016cd41a2a10e4ae"},"cell_type":"code","source":"val.loc[:, 'prediction'] = preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"126bf7ad2416f1d0ffa2f0a7de1da5eeccbc2cf8"},"cell_type":"code","source":"from sklearn.metrics import f1_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b741c38a247dcb894087d6333dbe4ef78bcbde5"},"cell_type":"code","source":"best_score = []\nts = np.arange(0, 1, 0.05)\nfor t in ts :\n    preds = np.where(val['prediction'] > t, 1, 0)\n    score =  f1_score(val['target'], preds )\n    print('Threshold: ', t , ',f1 score :', score )\n    best_score.append(score)\n    \nbest_idx = np.array(best_score).argmax()\n    \nprint('Best threshold :', ts[best_idx], 'Best Score :', best_score[best_idx] )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"35048975bf281e9176d894566aa5882659c88e03"},"cell_type":"markdown","source":"# TEST"},{"metadata":{"trusted":true,"_uuid":"ee39904f30818fc23fb648b5998b82b2c78d2faf"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\nm,r = get_mdl(train['target'])\npreds = np.zeros((len(test), 1))\npreds[:,0] = m.predict_proba(test_x.multiply(r))[:,1]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8c4f2d736c19ba531d69ef15fde7d71313741862"},"cell_type":"markdown","source":"# Submission"},{"metadata":{"trusted":true,"_uuid":"ac0f7cb5af9791fa79f765c17da6c32b1206fa03"},"cell_type":"code","source":"submission = pd.DataFrame({'qid': sub[\"qid\"]})\nsubmission['prediction'] = np.where(preds > ts[best_idx], 1 , 0)\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0fb4de643b6fd2d8a189108af0724f0d046ebb7e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}