{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"All stolen from https://www.kaggle.com/jhoward/nb-svm-strong-linear-baseline"},{"metadata":{"trusted":true,"_uuid":"1c83c5982d6891c9e669f87e03545a2288477cf7"},"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6711b26d587ebd4233af06f25daf592b2af2eeee"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\nsubm = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0980d4577806e74277ee85ede2e41008f954945d"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"01883681c1a85a083ff1083fed600eb9d014f8a3"},"cell_type":"code","source":"train['question_text'][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"913d084a8c357bb7e1dbb2b1b948292928234fe7"},"cell_type":"code","source":"lens = train.question_text.str.len()\nlens.mean(), lens.std(), lens.max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"605c784529817ff8de8482d80a5eee70bdfd7489"},"cell_type":"code","source":"lens.hist();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"457293b11b1aa076bc7ec44657255be0fbe38093"},"cell_type":"code","source":"len(train),len(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b69d5e35f0642f75b6994dc1b2fae307007668a1"},"cell_type":"code","source":"train['question_text'].fillna(\"unknown\", inplace=True)\ntest['question_text'].fillna(\"unknown\", inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b6bb7b52505a3ce2532d07e01eb5c457721fe32"},"cell_type":"code","source":"import re, string\nre_tok = re.compile(f'([{string.punctuation}“”¨«»®´·º½¾¿¡§£₤‘’])')\ndef tokenize(s): return re_tok.sub(r' \\1 ', s).split()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c8915cf60c8e1cfa2ad73ac751ded99f1acd3f5"},"cell_type":"code","source":"n = train.shape[0]\nvec = TfidfVectorizer(ngram_range=(1,2), tokenizer=tokenize,\n               min_df=3, max_df=0.9, strip_accents='unicode', use_idf=1,\n               smooth_idf=1, sublinear_tf=1 )\ntrn_term_doc = vec.fit_transform(train['question_text'])\ntest_term_doc = vec.transform(test['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f50c52c0218bdf5829f4da0499cc5be4e1eb29fb"},"cell_type":"code","source":"def pr(y_i, y):\n    p = x[y==y_i].sum(0)\n    return (p+1) / ((y==y_i).sum()+1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b3693b86732ab21fd54af253f0627f8ca3b779cd"},"cell_type":"code","source":"x = trn_term_doc\ntest_x = test_term_doc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"83da13d3cbe244b89fcf215a3b884447ed50993e"},"cell_type":"code","source":"def get_mdl(y):\n    y = y.values\n    r = np.log(pr(1,y) / pr(0,y))\n    m = LogisticRegression(C=4, dual=True)\n    x_nb = x.multiply(r)\n    return m.fit(x_nb, y), r","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ec0749a528b9e947a36318502c7add27405d327"},"cell_type":"code","source":"m,r = get_mdl(train['target'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b159b417f4e42d941696785f768130cc59112636"},"cell_type":"code","source":"preds = m.predict_proba(test_x.multiply(r))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"220a5eb2a3340d56a5a2a425ad5a3ca256de238f"},"cell_type":"code","source":"preds = preds[:,1] ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb098257184de4ea97f5402d1dda142b0a4cba81"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a614823f9f44f9afbff8132f63de0519da51eaae"},"cell_type":"code","source":"thresholds = np.linspace(0, 1, 1000)\nscore = 0.0\ntest_threshold=0.5\nbest_threshold=np.zeros(1)\nbest_val = np.zeros(1)\n\nfor threshold in thresholds:\n    test_threshold = threshold\n    max_val = np.max(val_pred[:,0])\n    val_predict = (val_pred[:,0] > test_threshold)\n    score = f1_score(y_val, val_predict)\n    if score > best_val:\n        best_threshold = threshold\n        best_val = score\n\nprint(\"Threshold %0.6f, F1: %0.6f\" % (best_threshold,best_val))\ntest_threshold = best_threshold\n\nprint(\"Best threshold: \")\nprint(best_threshold)\nprint(\"Best f1:\")\nprint(best_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44120fcb6c28f4f9b1a352a120ce241e408533cd"},"cell_type":"code","source":"y_te = (preds > best_threshold).astype(np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87067a959271ac735e06e06d51fca9645be18d1d"},"cell_type":"code","source":"submit_df = pd.DataFrame({\"qid\": test[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}