{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nfrom sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer\nfrom sklearn.model_selection import train_test_split, cross_val_score\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true},"cell_type":"code","source":"train_df=pd.read_csv('../input/train.csv')[:200000]\ntrain_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5b6a8b32653b5cb17affb5e3daecd0363512a29"},"cell_type":"code","source":"vectorizer = TfidfVectorizer(min_df=5,strip_accents='unicode',lowercase =True, analyzer='word',use_idf=True, smooth_idf=True, sublinear_tf=True, \n                        stop_words = 'english',tokenizer=word_tokenize)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d8b86b067e8c8ffe56913e80f8699fd329a6cf78"},"cell_type":"markdown","source":"# split in words"},{"metadata":{"trusted":true,"_uuid":"443e5c8a440a29ee2dfb823ac241f21bb1791e51"},"cell_type":"code","source":"#train_vectorized = vectorizer.transform(train_df.question_text.values)\n\n#train1_tfidf=\nvectorizer.fit_transform(train_df[train_df.target==1].question_text.values)\nwoord1=vectorizer.get_feature_names()\n#train0_tfidf=\nvectorizer.fit_transform(train_df[train_df.target==0].question_text.values)\nwoord0=vectorizer.get_feature_names()\n\nwoord01=[x for x in set(woord1) if x in set(woord0)]\nprint(len(woord01),len(woord1),len(woord0))\nwordsimpf=[x for x in set(woord1) if x not in set(woord01)]\nlen(wordsimpf)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"c0325948f965d608e346668e88e1eac46f4dae80"},"cell_type":"markdown","source":"# take glove vh matrix"},{"metadata":{"trusted":true,"_uuid":"0afc8400b76d05d3a51de10f9f386483c2834042"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"242f93543d37dbec8c6330c2b9321ff625a12d9b"},"cell_type":"code","source":"\ndef embedword(word_index,word_pos,word_neg):\n    nb_words = min(60000, len(word_index))\n    embedding_matrix_1 =pd.DataFrame([])\n    i=0\n    for word in word_index:\n        embedding_vector = embeddings_index.get(word)\n        embedding_matrix_1=embedding_matrix_1.append(pd.DataFrame(embedding_vector,columns=[i]).T)\n        i=i+1\n\n\n    #el embeddings_index\n    embedding_matrix_1['woord']=word_index\n    embedding_matrix_1['target']=0\n    for w in word_pos:\n        pos=embedding_matrix_1.loc[embedding_matrix_1['woord'] ==w].index\n        if pos.size>0:\n            embedding_matrix_1.iat[pos[0],301]=1\n\n    for w in word_neg:\n        pos=embedding_matrix_1.loc[embedding_matrix_1['woord'] ==w].index\n        if pos.size>0:\n            embedding_matrix_1.iat[pos[0],301]=2\n    embedding_matrix_1.plot.scatter(x=0,y=1,c='target',colormap='viridis')\n    return embedding_matrix_1.fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d4ca65553e9953fa66ac208ecbb12506e67aea6","scrolled":true},"cell_type":"code","source":"wh1=embedword(woord1,woord0,wordsimpf)\nwh0=embedword(woord0,woord1,wordsimpf)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"adcf76b58a36a0c9b2cda42622e5113b8e8b82c1"},"cell_type":"code","source":"del embeddings_index,EMBEDDING_FILE","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"421b22ceeb1ed58d75b99ae8c98409d8a7ecacb7"},"cell_type":"markdown","source":"# classify words with the bad words"},{"metadata":{"trusted":true,"_uuid":"b347942bcd3156b0aee1bddf1b7d2c7cd0f5dc4b"},"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom xgboost import XGBClassifier\n\nclf=KNeighborsClassifier(3)\nclf = SGDClassifier(max_iter=1000)\nclf =XGBClassifier(max_depth=5, base_score=0.005)\nwht=wh0.append(wh1)\nclf.fit(wht.drop(labels=['woord','target'],axis=1).fillna(0),wht['target'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16dfe616d76ffb3c5a637567a094161884297de7"},"cell_type":"code","source":"wht['pred']=clf.predict(wht.drop(labels=['woord','target'],axis=1).fillna(0))  #==wht['target']).mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a2f4faa5c6fb95f5df8e870cfe8ab6ed0c199cc"},"cell_type":"code","source":"wb=wht[wht[\"pred\"]>0].woord\n\nvectorizer2 = TfidfVectorizer(min_df=5,strip_accents='unicode',lowercase =True, analyzer='word',use_idf=True, smooth_idf=True, sublinear_tf=True, \n                        stop_words = 'english',vocabulary=np.unique(wb),tokenizer=word_tokenize)\n#vectorizer2 = CountVectorizer(min_df=5,strip_accents='unicode',lowercase =True, analyzer='word',stop_words = 'english',vocabulary=np.unique(wb),tokenizer=word_tokenize)\ntrain_tfidf=vectorizer2.fit_transform([train_df.question_text.sum()] )\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e042fa800fcfcd292e4c71924c2c73c6719c0f61"},"cell_type":"code","source":"len( vectorizer2.get_feature_names() )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e23d8d2010aa5e5c1abf7876cec92342dbcd4dc5"},"cell_type":"markdown","source":"# shitty words"},{"metadata":{"trusted":true,"_uuid":"0b278db39889bac12a3f85c1a40ea0f8ecb7110c"},"cell_type":"code","source":"wordcount=pd.DataFrame(vectorizer2.get_feature_names(),columns=['word'])\nwordcount['idf']=1/train_tfidf.T.sum(axis=1)\ntemp=wht[wht[\"pred\"]>0].groupby(\"woord\").max()\nwordcount['target']=temp.target.values\nwordcount['pred']=temp.pred.values\nwordcount[  wordcount.target>1].sort_values(by=['idf','target','word'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"95feb12bcc48673a09c9c7452f9d035a5f5965ce"},"cell_type":"code","source":"wordcount[  wordcount.pred>1].sort_values(by=['idf','target','word'])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}