{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"941ce7c6e0509188fbe9508ec8426aea256d2062"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction import DictVectorizer\nfrom sklearn.linear_model import Ridge\nfrom scipy.sparse import hstack\nfrom nltk.corpus import stopwords\nstop = stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"input = pd.read_csv('../input/train.csv')\ntest= pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d55a57dcd8eb0473e3e812497f5a831cdf22dc5d"},"cell_type":"code","source":"input.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25a208d79b4187cc9f9064cdc0795aea70a5b34c"},"cell_type":"code","source":"input.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18d381ffb71636230168c88c626bad1f2fb6b62c"},"cell_type":"code","source":"#from sklearn.model_selection import train_test_split\n#trainX,testX,trainY,testY = train_test_split(input.drop(['target'],axis=1), input['target'], test_size=0.33, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1226f2ab293b57be27218c13be6f74e08ea07265"},"cell_type":"code","source":"trainX= input.drop(['target'],axis=1)\ntrainY= input['target']\ntestX= test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b89cc5b8a57899071d9edf88e2e5cf72f756051a"},"cell_type":"code","source":"def tokenizer(text):\n    if text:\n        result = re.findall('[a-z]{2,}', text.lower())\n    else:\n        result = []\n    return result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"094f8a036c9e83efe3649adcbe7d2383acd35449"},"cell_type":"code","source":"testX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8f707fab3682267c7cd22ff6a9752ff6018f3d5"},"cell_type":"code","source":"trainX['word_count'] = trainX['question_text'].apply(lambda x: len(str(x).split(\" \")))\ntestX['word_count'] = testX['question_text'].apply(lambda x: len(str(x).split(\" \")))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8f00c75b386d63faee36fa09a2498841d95e107"},"cell_type":"code","source":"trainX['char_count'] = trainX['question_text'].str.len() ## this also includes spaces\ntestX['char_count'] = testX['question_text'].str.len() \ntrainX[['question_text','char_count']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5647731c47493bb900c5a7dceb98870d08eaa66f"},"cell_type":"code","source":"def avg_word(sentence):\n  words = sentence.split()\n  return (sum(len(word) for word in words)/len(words))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7136f2dd30fc17f3afcbd594fe006fbf8b4648a4"},"cell_type":"code","source":"trainX['avg_word'] = trainX['question_text'].apply(lambda x: avg_word(x))\ntestX['avg_word'] = testX['question_text'].apply(lambda x: avg_word(x))\ntrainX[['question_text','avg_word']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6095379506381e89a8bdc14582f7c8c6a322b74"},"cell_type":"code","source":"trainX['stopwords'] = trainX['question_text'].apply(lambda x: len([x for x in x.split() if x in stop]))\ntestX['stopwords'] = testX['question_text'].apply(lambda x: len([x for x in x.split() if x in stop]))\ntrainX[['question_text','stopwords']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3dfa3fa2fd4ba09d0ed48c42e31791fe08d0740"},"cell_type":"code","source":"trainX['numerics'] = trainX['question_text'].apply(lambda x: len([x for x in x.split() if x.isdigit()]))\ntestX['numerics'] = testX['question_text'].apply(lambda x: len([x for x in x.split() if x.isdigit()]))\ntrainX[['question_text','numerics']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d99e25a9c89113c916548b2a35828a61aa9f46b"},"cell_type":"code","source":"trainX['upper'] = trainX['question_text'].apply(lambda x: len([x for x in x.split() if x.isupper()]))\ntestX['upper'] = testX['question_text'].apply(lambda x: len([x for x in x.split() if x.isupper()]))\ntrainX[['question_text','upper']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99a1cdb6738caa741146a036a0aa209d3558a4ed"},"cell_type":"code","source":"trainX['qmark']=trainX[['question_text']].applymap(lambda x: str.count(x, '?'))\ntrainX['esclampoint']=trainX[['question_text']].applymap(lambda x: str.count(x, '!'))\ntrainX['atrateof']=trainX[['question_text']].applymap(lambda x: str.count(x, '@'))\ntestX['qmark']=testX[['question_text']].applymap(lambda x: str.count(x, '?'))\ntestX['esclampoint']=testX[['question_text']].applymap(lambda x: str.count(x, '!'))\ntestX['atrateof']=testX[['question_text']].applymap(lambda x: str.count(x, '@'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3bb7da2201fe4870f1aad1fbde07004c24bae4bb"},"cell_type":"code","source":"trainX1 = trainX['question_text']\ntestX1 = testX['question_text']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4e985eba7f62653572fce9557761e13f3a13df12"},"cell_type":"code","source":"import time\nimport re\nvect = TfidfVectorizer(tokenizer=tokenizer, stop_words='english',max_features=5000)\nstart = time.time()\nX_train_vect = vect.fit_transform(trainX1)\nX_test_vect = vect.transform(testX1)\nend = time.time()\nprint('Time to train vectorizer and transform training text: %0.2fs' % (end - start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ed3ef229e6e0d77edf3a5362177f70a8defdd0c"},"cell_type":"code","source":"X_train_vect","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d4b07388f74d02c5a676677dc74006e8c3ba5c2"},"cell_type":"code","source":"trainX.head().T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75caf513743216d0f9f2d3091623d44c03c5b902"},"cell_type":"code","source":"import scipy\nsparse_matrix= scipy.sparse.csr_matrix(trainX.drop(['qid','question_text'],axis=1))\nsparse_matrix1= scipy.sparse.csr_matrix(testX.drop(['qid','question_text'],axis=1))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0842364fc87888c82d3f5648491f4214e4fa47dd"},"cell_type":"code","source":"new_features = scipy.sparse.hstack((X_train_vect,sparse_matrix))\ntest_new_features = scipy.sparse.hstack((X_test_vect,sparse_matrix1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cd28174307046afd267d96de24b6709225ad6a0"},"cell_type":"code","source":"import xgboost as xgb\n#model = SGDRegressor(loss='squared_loss', penalty='l2', random_state=seed, max_iter=5)\n#sgd = SGDClassifier(loss=\"hinge\", penalty=\"l2\")\nstart = time.time()\ngbm = xgb.XGBClassifier(max_depth=200, n_estimators=100, learning_rate=0.1,silent=False,n_jobs=-1).fit(new_features, trainY)#,\n                                                                                    #       eval_set=[(test_new_features, testY)],\n                                                                                    #       eval_metric = ['logloss', 'auc'],early_stopping_rounds=25,\n                                                                                    #       verbose=True)\nend = time.time()\nprint('Time to train model: %0.2fs' % (end -start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"313cede901b256297041488f2a873c5705bba4e5"},"cell_type":"markdown","source":"\nimport xgboost as xgb\n#model = SGDRegressor(loss='squared_loss', penalty='l2', random_state=seed, max_iter=5)\n#sgd = SGDClassifier(loss=\"hinge\", penalty=\"l2\")\nstart = time.time()\ngbm = xgb.XGBClassifier(max_depth=200, n_estimators=10, \n                        learning_rate=0.1,silent=False,\n                        n_jobs=-1).fit(new_features, trainY, \n                                       eval_set=[(test_new_features, testY)], \n                                       eval_metric = ['logloss', 'auc'],\n                                       early_stopping_rounds=5,verbose=True)\nend = time.time()\nprint('Time to train model: %0.2fs' % (end -start))"},{"metadata":{"trusted":true,"_uuid":"170ea1471e247e9ee01f8e74c44e147715b0f037"},"cell_type":"code","source":"model = gbm.best_iteration","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eb91b2567eecc46de6f91d2d6639cdfeb2f020b6"},"cell_type":"code","source":"y_pred = gbm.predict(test_new_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5ce30f5c1048ca1c47bc5aa43ec767b001f8230"},"cell_type":"code","source":"from sklearn.metrics import f1_score\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b21ad2e7d04c8dc109406bc08265a90870d62d2c"},"cell_type":"code","source":"#f1_score(y_pred,testY)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd4f1de5e0f34ad795df6990a0e06e957fe60cb3"},"cell_type":"code","source":"test['prediction']= pd.DataFrame(y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6fe57103dabd8e96f4b5383a1f2a137e65c7dbbc"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ac7ac78b7362a358fbb4bccb7571b132758f60e"},"cell_type":"code","source":"test[['qid','prediction']].to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5573951fa4661e9a0ea0d3b3347286d3a0870545"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}