{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25e890e1e4a1098633109f57889e2f8ea0da5752"},"cell_type":"code","source":"## Using simple TFIDF features and SVM \n## Observation -- Although accuracy comes out to be high, that is not a good measure-the F1 score is low","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8678dc68e7b4fad4d16a5975a1aed0cf0dda363a"},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbaf0be04b9eab465abb80634377dfce58e229d3"},"cell_type":"code","source":"train_df.tail(30)\ninsincere_q = train_df[train_df['target']==1]\ninsincere_q.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea9aaa3bd29e15aad84bfb26ab589d3d5caa133a"},"cell_type":"code","source":"# function to clean data\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer\nimport re\nfrom nltk.stem import WordNetLemmatizer\n\nlemm_ = WordNetLemmatizer()\nst = PorterStemmer()\nstops = set(stopwords.words(\"english\"))\ndef cleanData(text, lowercase = True, remove_stops = True, stemming = False, lemma = True):\n    #txt = str(text)\n    #print(text)\n    #txt = text.encode('utf-8').strip()\n    txt = str(text)\n    txt = re.sub(r'[^a-zA-Z. ]+|(?<=\\\\d)\\\\s*(?=\\\\d)|(?<=\\\\D)\\\\s*(?=\\\\d)|(?<=\\\\d)\\\\s*(?=\\\\D)',r'',txt)\n    txt = re.sub(r'\\n',r' ',txt)\n    \n    #converting to lower case\n    if lowercase:\n        txt = \" \".join([w.lower() for w in txt.split()])\n    \n    # removing stop words\n    if remove_stops:\n        txt = \" \".join([w for w in txt.split() if w not in stops])\n    \n    # stemming\n    if stemming:\n        txt = \" \".join([st.stem(w) for w in txt.split()])\n        \n    if lemma:\n        txt = \" \".join([lemm_.lemmatize(w) for w in txt.split()])\n\n    return txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc868c0e48a23cb6a58651e3d301150ab798b710"},"cell_type":"code","source":"train_df['clean_question_text'] = train_df['question_text'].map(lambda x: cleanData(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c495f1c0d0473d75d4188dec158c414af1ad12bd"},"cell_type":"code","source":"test_df['clean_question_text'] = test_df['question_text'].map(lambda x: cleanData(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de40c0d62fc93618d5edc047786c2311709ca232"},"cell_type":"code","source":"max_features = 50000  ##More than this would filter in noise also\ntfidf_vectorizer = TfidfVectorizer(ngram_range =(2,4) , max_df=0.90, min_df=5, max_features=max_features) ##4828 features found\n#tfidf_feature_names = tfidf_vectorizer.get_feature_names()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d39d0068873f5119f3d574a8271c74eb11063136"},"cell_type":"code","source":"X = tfidf_vectorizer.fit_transform(train_df['clean_question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae16e736b3fd0da6dad3acc43b3d871f0ba0797e"},"cell_type":"code","source":"X_te = tfidf_vectorizer.transform(test_df['clean_question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9287dd180153a5cd8876fb3bd2a5749c7aceacf5"},"cell_type":"code","source":"tfidf_feature_names = tfidf_vectorizer.get_feature_names()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"429e7890a5f75b04334fda8058960d716f4854df"},"cell_type":"code","source":"y = train_df[\"target\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48035d714cb44a9ce81f3bde142ebbdb0374a8d3"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.3,random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4069af28af56c51f8865356e737a87fcbe3df28"},"cell_type":"code","source":"# Classification and prediction\nclf = LogisticRegression(C=10, penalty='l1')\nclf.fit(X_train, y_train)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44537efddb9c8a62d09c8e657adc0c3ae272b560"},"cell_type":"code","source":"clf.score(X_val, y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"705d1ab753628b72a640deaa9c8798ce35eed640"},"cell_type":"code","source":"p_test = clf.predict_proba(X_te)[:, 0]\ny_te = (p_test > 0.5).astype(np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d32a756c054b1de8649850e58059913d814921e"},"cell_type":"code","source":"from sklearn.svm import LinearSVC\nsvm_model = LinearSVC(C=0.5).fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99710adc54c74aa2d95c10e988746e7241c2d4da"},"cell_type":"code","source":"score = svm_model.score(X_train, y_train)\nprint('score', score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"50765203d25a3a9e5c9d9587bf1f7f7e1ccd9e05"},"cell_type":"code","source":"#pred_test_y = (pred_test_y > best_thresh).astype(int)\npred_test_y = svm_model.predict(X_te)\n\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4b571f49d9def25ae08c0ddd8fb1a68165c9e12"},"cell_type":"code","source":"out_df.tail()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78ad506df14347fe1de8a2cc35ecc5275585c8a8"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}